From 7efccd12d5d4f0f090fa334b840cba1a39681245 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Mon, 25 May 2026 15:47:24 +0200 Subject: [PATCH 01/11] infra: add MPLS nexthop type and address family MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Introduce the basic infrastructure types needed for MPLS support. Add GR_NH_T_MPLS to the nexthop type enum for MPLS label imposition and swap operations. Add GR_AF_MPLS (AF_MPLS = 28) to the address family enum so that MPLS can register per-AF operations such as nexthop resolution callbacks and per-VRF FIB lifecycle management. Reserve a fib_mpls slot in the private VRF struct for the per-VRF label forwarding table that will be populated by the MPLS module. Extend nexthop_af_ops_from_mbuf() to recognize MPLS packets held for ARP resolution. Packets marked with RTE_PTYPE_TUNNEL_MPLS_IN_GRE are dispatched to the MPLS AF ops so that they get resubmitted to the correct datapath node after the gateway nexthop becomes reachable. Signed-off-by: Matej Mužila --- api/gr_net_types.h | 4 ++++ modules/infra/api/gr_nexthop.h | 3 +++ modules/infra/control/l3_nexthop.c | 6 +++++ modules/infra/control/nexthop.c | 1 + modules/infra/control/vrf.h | 1 + modules/ip/datapath/ip_error.c | 10 +++++++- modules/ip6/datapath/ip6_error.c | 38 ++++++++++++++++++++++++++++++ modules/ip6/datapath/ip6_output.c | 6 ++--- modules/l2/control/vxlan.c | 1 + 9 files changed, 66 insertions(+), 4 deletions(-) diff --git a/api/gr_net_types.h b/api/gr_net_types.h index efaf0c830..079bb9566 100644 --- a/api/gr_net_types.h +++ b/api/gr_net_types.h @@ -27,6 +27,7 @@ typedef enum : uint8_t { GR_AF_UNSPEC = AF_UNSPEC, GR_AF_IP4 = AF_INET, GR_AF_IP6 = AF_INET6, + GR_AF_MPLS = AF_MPLS, } addr_family_t; // Convert address family enum to string representation. @@ -38,6 +39,8 @@ static inline const char *gr_af_name(addr_family_t af) { return "ipv4"; case GR_AF_IP6: return "ipv6"; + case GR_AF_MPLS: + return "mpls"; } return "?"; } @@ -48,6 +51,7 @@ static inline bool gr_af_valid(addr_family_t af) { case GR_AF_UNSPEC: case GR_AF_IP4: case GR_AF_IP6: + case GR_AF_MPLS: return true; } return false; diff --git a/modules/infra/api/gr_nexthop.h b/modules/infra/api/gr_nexthop.h index ef9497914..e3f7ffcab 100644 --- a/modules/infra/api/gr_nexthop.h +++ b/modules/infra/api/gr_nexthop.h @@ -38,6 +38,7 @@ typedef enum : uint8_t { GR_NH_T_BLACKHOLE, // Drop packets silently. GR_NH_T_REJECT, // Drop packets with ICMP error. GR_NH_T_GROUP, // ECMP for multipath routing. + GR_NH_T_MPLS, // MPLS label imposition/swap. #define GR_NH_T_ALL UINT8_C(0xff) // Match all types in list operations. } gr_nh_type_t; @@ -199,6 +200,8 @@ static inline const char *gr_nh_type_name(const gr_nh_type_t type) { return "reject"; case GR_NH_T_GROUP: return "group"; + case GR_NH_T_MPLS: + return "MPLS"; } return "?"; } diff --git a/modules/infra/control/l3_nexthop.c b/modules/infra/control/l3_nexthop.c index 6944839b7..c4892ad31 100644 --- a/modules/infra/control/l3_nexthop.c +++ b/modules/infra/control/l3_nexthop.c @@ -45,6 +45,8 @@ const struct nexthop_af_ops *nexthop_af_ops_from_mbuf(const struct rte_mbuf *m) return af_ops[GR_AF_IP4]; if (m->packet_type & RTE_PTYPE_L3_IPV6) return af_ops[GR_AF_IP6]; + if ((m->packet_type & RTE_PTYPE_TUNNEL_MASK) == RTE_PTYPE_TUNNEL_MPLS_IN_GRE) + return af_ops[GR_AF_MPLS]; return NULL; } @@ -78,6 +80,9 @@ static inline void set_nexthop_key( key->ipv6.a[3] = iface_id & 0xff; } break; + case GR_AF_MPLS: + ABORT("AF_MPLS has no nexthop key with gw"); + break; case GR_AF_UNSPEC: ABORT("AF_UNSPEC has no nexthop key with gw"); break; @@ -206,6 +211,7 @@ static bool l3_equal(const struct nexthop *a, const struct nexthop *b) { case GR_AF_IP6: return rte_ipv6_addr_eq(&l3_a->ipv6, &l3_b->ipv6); case GR_AF_UNSPEC: + case GR_AF_MPLS: return true; } return false; diff --git a/modules/infra/control/nexthop.c b/modules/infra/control/nexthop.c index d4549f39e..6a33a5010 100644 --- a/modules/infra/control/nexthop.c +++ b/modules/infra/control/nexthop.c @@ -240,6 +240,7 @@ bool nexthop_type_valid(gr_nh_type_t type) { case GR_NH_T_BLACKHOLE: case GR_NH_T_REJECT: case GR_NH_T_GROUP: + case GR_NH_T_MPLS: return true; } return false; diff --git a/modules/infra/control/vrf.h b/modules/infra/control/vrf.h index 428ea6e34..734d2214a 100644 --- a/modules/infra/control/vrf.h +++ b/modules/infra/control/vrf.h @@ -13,6 +13,7 @@ GR_IFACE_INFO(GR_IFACE_TYPE_VRF, iface_info_vrf, { uint32_t vrf_ifindex; void *fib4; void *fib6; + void *fib_mpls; }); // Map a grout VRF ID to the kernel routing table ID. diff --git a/modules/ip/datapath/ip_error.c b/modules/ip/datapath/ip_error.c index d4cb27233..b5ff96464 100644 --- a/modules/ip/datapath/ip_error.c +++ b/modules/ip/datapath/ip_error.c @@ -84,7 +84,15 @@ ip_error_process(struct rte_graph *graph, struct rte_node *node, void **objs, ui icmp->icmp_code = ctx->icmp_code; icmp->icmp_cksum = 0; icmp->icmp_ident = 0; - icmp->icmp_seq_nb = 0; + if (ctx->icmp_code == RTE_ICMP_CODE_UNREACH_FRAG) { + // RFC 1191: next-hop MTU in the seq_nb field position + const struct iface *err_iface = mbuf_data(mbuf)->iface; + icmp->icmp_seq_nb = (err_iface != NULL) ? + rte_cpu_to_be_16(err_iface->mtu) : + 0; + } else { + icmp->icmp_seq_nb = 0; + } edge = ICMP_OUTPUT; next: diff --git a/modules/ip6/datapath/ip6_error.c b/modules/ip6/datapath/ip6_error.c index 16fde4772..777293650 100644 --- a/modules/ip6/datapath/ip6_error.c +++ b/modules/ip6/datapath/ip6_error.c @@ -26,10 +26,12 @@ enum edges { static uint16_t ip6_error_process(struct rte_graph *graph, struct rte_node *node, void **objs, uint16_t nb_objs) { const struct ip6_error_ctx *ctx = ip6_error_ctx(node); + struct icmp6_err_pkt_too_big *ptb; struct icmp6_err_dest_unreach *du; struct icmp6_err_ttl_exceeded *te; const struct nexthop_info_l3 *l3; struct ip6_local_mbuf_data *d; + const struct iface *err_iface; const struct iface *iface; const struct nexthop *nh; struct rte_ipv6_hdr *ip; @@ -69,6 +71,15 @@ ip6_error_process(struct rte_graph *graph, struct rte_node *node, void **objs, u goto next; } break; + case ICMP6_ERR_PKT_TOO_BIG: + ptb = gr_mbuf_prepend(mbuf, ptb); + if (unlikely(ptb == NULL)) { + edge = NO_HEADROOM; + goto next; + } + err_iface = mbuf_data(mbuf)->iface; + ptb->mtu = (err_iface != NULL) ? rte_cpu_to_be_32(err_iface->mtu) : 0; + break; default: ABORT("unexpected icmp_type value %hhu", ctx->icmp_type); break; @@ -120,6 +131,15 @@ static int no_route_init(const struct rte_graph *, struct rte_node *node) { return 0; } +static int pkt_too_big_init(const struct rte_graph *, struct rte_node *node) { + struct ip6_error_ctx *ctx; + + ctx = ip6_error_ctx(node); + ctx->icmp_type = ICMP6_ERR_PKT_TOO_BIG; + ctx->icmp_code = 0; + return 0; +} + static struct rte_node_register dest_unreach_node = { .name = "ip6_error_dest_unreach", .process = ip6_error_process, @@ -144,6 +164,18 @@ static struct rte_node_register ttl_exceeded_node = { .init = ttl_exceeded_init, }; +static struct rte_node_register pkt_too_big_node = { + .name = "ip6_error_pkt_too_big", + .process = ip6_error_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp6_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, + .init = pkt_too_big_init, +}; + static struct gr_node_info dest_unreach_info = { .node = &dest_unreach_node, .type = GR_NODE_T_L3, @@ -154,5 +186,11 @@ static struct gr_node_info ttl_exceeded_info = { .type = GR_NODE_T_L3, }; +static struct gr_node_info pkt_too_big_info = { + .node = &pkt_too_big_node, + .type = GR_NODE_T_L3, +}; + GR_NODE_REGISTER(dest_unreach_info); GR_NODE_REGISTER(ttl_exceeded_info); +GR_NODE_REGISTER(pkt_too_big_info); diff --git a/modules/ip6/datapath/ip6_output.c b/modules/ip6/datapath/ip6_output.c index 4853fc3cc..8e0972551 100644 --- a/modules/ip6/datapath/ip6_output.c +++ b/modules/ip6/datapath/ip6_output.c @@ -94,6 +94,8 @@ ip6_output_process(struct rte_graph *graph, struct rte_node *node, void **objs, goto next; } + mbuf_data(mbuf)->iface = iface; + if (rte_pktmbuf_pkt_len(mbuf) > iface->mtu) { edge = TOO_BIG; goto next; @@ -102,7 +104,6 @@ ip6_output_process(struct rte_graph *graph, struct rte_node *node, void **objs, // Determine what is the next node based on the output interface type // By default, it will be eth_output unless another output node was registered. edge = iface_type_edges[iface->type]; - mbuf_data(mbuf)->iface = iface; if (edge != ETH_OUTPUT) goto next; @@ -157,7 +158,7 @@ static struct rte_node_register output_node = { [HOLD] = "ip6_hold", [ERROR] = "ip6_output_error", [DEST_UNREACH] = "ip6_error_dest_unreach", - [TOO_BIG] = "ip6_output_too_big", + [TOO_BIG] = "ip6_error_pkt_too_big", }, }; @@ -171,4 +172,3 @@ static struct gr_node_info info = { GR_NODE_REGISTER(info); GR_DROP_REGISTER(ip6_output_error); -GR_DROP_REGISTER(ip6_output_too_big); diff --git a/modules/l2/control/vxlan.c b/modules/l2/control/vxlan.c index 3d2a7367b..8005b9b31 100644 --- a/modules/l2/control/vxlan.c +++ b/modules/l2/control/vxlan.c @@ -207,6 +207,7 @@ static int iface_vxlan_reconfig( cur->template.ipv6.vxlan.vx_vni = vxlan_encode_vni(cur->vni); break; case GR_AF_UNSPEC: + case GR_AF_MPLS: break; } From 25ba6b01810316b416850052dfcad811c9744732 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Mon, 27 Jul 2026 10:57:41 +0200 Subject: [PATCH 02/11] infra: move checksum helpers to generic header MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extract fixup_checksum_16 and fixup_checksum_32 from nat_datapath.h into a standalone checksum.h so they can be reused outside of the NAT module. nat_datapath.h now includes the new header. Signed-off-by: Matej Mužila --- modules/infra/datapath/checksum.h | 33 ++++++++++++++++++++++++++ modules/policy/datapath/nat_datapath.h | 30 +---------------------- 2 files changed, 34 insertions(+), 29 deletions(-) create mode 100644 modules/infra/datapath/checksum.h diff --git a/modules/infra/datapath/checksum.h b/modules/infra/datapath/checksum.h new file mode 100644 index 000000000..71387f560 --- /dev/null +++ b/modules/infra/datapath/checksum.h @@ -0,0 +1,33 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#pragma once + +#include + +// RFC 1624 incremental checksum update for a 16-bit field. +static inline rte_be16_t +fixup_checksum_16(rte_be16_t old_cksum, rte_be16_t old_field, rte_be16_t new_field) { + uint32_t sum; + + sum = ~old_cksum & 0xffff; + sum += (~old_field & 0xffff) + new_field; + sum = (sum >> 16) + (sum & 0xffff); + sum += (sum >> 16); + + return ~sum & 0xffff; +} + +// RFC 1624 incremental checksum update for a 32-bit field. +static inline rte_be16_t +fixup_checksum_32(rte_be16_t old_cksum, ip4_addr_t old_addr, ip4_addr_t new_addr) { + uint32_t sum; + + sum = ~old_cksum & 0xffff; + sum += (~old_addr & 0xffff) + (new_addr & 0xffff); + sum += (~old_addr >> 16) + (new_addr >> 16); + sum = (sum >> 16) + (sum & 0xffff); + sum += (sum >> 16); + + return ~sum & 0xffff; +} diff --git a/modules/policy/datapath/nat_datapath.h b/modules/policy/datapath/nat_datapath.h index 1e791d46d..8d0319740 100644 --- a/modules/policy/datapath/nat_datapath.h +++ b/modules/policy/datapath/nat_datapath.h @@ -3,6 +3,7 @@ #pragma once +#include "checksum.h" #include "iface.h" #include "nexthop.h" @@ -16,35 +17,6 @@ GR_NH_TYPE_INFO(GR_NH_T_DNAT, nexthop_info_dnat, { struct nexthop *arp; }); -static inline rte_be16_t -fixup_checksum_16(rte_be16_t old_cksum, rte_be16_t old_field, rte_be16_t new_field) { - uint32_t sum; - - // RFC 1624: HC' = ~(~HC + ~m + m') - // Note: 1's complement sum is endian-independent (RFC 1071, page 2). - sum = ~old_cksum & 0xffff; - sum += (~old_field & 0xffff) + new_field; - sum = (sum >> 16) + (sum & 0xffff); - sum += (sum >> 16); - - return ~sum & 0xffff; -} - -static inline rte_be16_t -fixup_checksum_32(rte_be16_t old_cksum, ip4_addr_t old_addr, ip4_addr_t new_addr) { - uint32_t sum; - - // Checksum 32-bit datum as as two 16-bit. Note, the first - // 32->16 bit reduction is not necessary. - sum = ~old_cksum & 0xffff; - sum += (~old_addr & 0xffff) + (new_addr & 0xffff); - sum += (~old_addr >> 16) + (new_addr >> 16); - sum = (sum >> 16) + (sum & 0xffff); - sum += (sum >> 16); - - return ~sum & 0xffff; -} - typedef enum { NAT_VERDICT_CONTINUE, NAT_VERDICT_FINAL, From 5630fa5106c790d7b3f307b4bf04ba0e9cb89729 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Mon, 25 May 2026 16:05:57 +0200 Subject: [PATCH 03/11] mpls: add public API header MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The nexthop info struct carries the output label stack, initial TTL, a gateway address for ARP/NDP resolution, and an optional payload type for egress label disposition. MPLS nexthops are managed through the generic GR_NH_ADD/DEL API, same as SRv6. Label routes map an incoming label to a nexthop in the per-VRF LFIB. The API provides add, delete, get, and streaming list operations with the corresponding event types for route change notifications. Signed-off-by: Matej Mužila --- modules/mpls/api/gr_mpls.h | 101 +++++++++++++++++++++++++++++++++++++ 1 file changed, 101 insertions(+) create mode 100644 modules/mpls/api/gr_mpls.h diff --git a/modules/mpls/api/gr_mpls.h b/modules/mpls/api/gr_mpls.h new file mode 100644 index 000000000..a5b590516 --- /dev/null +++ b/modules/mpls/api/gr_mpls.h @@ -0,0 +1,101 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#pragma once + +#include +#include +#include + +#include + +#define GR_MPLS_MODULE 0xf00e + +#define GR_MPLS_MAX_LABELS 16 + +#define GR_MPLS_MAX_STACK_DEPTH 30 + +#define GR_MPLS_LABEL_MAX 0xFFFFF + +// Reserved MPLS label values (RFC 3032). +enum gr_mpls_reserved_labels : uint32_t { + GR_MPLS_LABEL_IPV4_EXPLICIT_NULL = 0, + GR_MPLS_LABEL_ROUTER_ALERT = 1, + GR_MPLS_LABEL_IPV6_EXPLICIT_NULL = 2, + GR_MPLS_LABEL_IMPLICIT_NULL = 3, + GR_MPLS_LABEL_FIRST_UNRESERVED = 16, +}; + +struct gr_nexthop_info_mpls { + uint8_t n_labels; + uint8_t ttl; // 0 = copy from payload. + addr_family_t payload_af; // Payload type after pop (GR_AF_UNSPEC = auto-detect). + struct l3_addr via; + uint32_t labels[GR_MPLS_MAX_LABELS]; +}; + +// label routes +enum gr_mpls_requests : uint32_t { + GR_MPLS_LABEL_ROUTE_ADD = GR_MSG_TYPE(GR_MPLS_MODULE, 0x0001), + GR_MPLS_LABEL_ROUTE_DEL, + GR_MPLS_LABEL_ROUTE_GET, + GR_MPLS_LABEL_ROUTE_LIST, +}; + +// MPLS label route entry. +struct gr_mpls_label_route { + uint16_t vrf_id; + uint32_t in_label; + uint32_t nh_id; + gr_nh_origin_t origin; +}; + +// Add a label route to the LFIB. +struct gr_mpls_label_route_add_req { + uint16_t vrf_id; + uint32_t in_label; // 0 to GR_MPLS_LABEL_MAX. + uint32_t nh_id; // Must reference a GR_NH_T_MPLS nexthop. + gr_nh_origin_t origin; + uint8_t exist_ok; +}; + +GR_REQ(GR_MPLS_LABEL_ROUTE_ADD, struct gr_mpls_label_route_add_req, struct gr_empty); + +// Delete a label route from the LFIB. +struct gr_mpls_label_route_del_req { + uint16_t vrf_id; + uint32_t in_label; + uint8_t missing_ok; +}; + +GR_REQ(GR_MPLS_LABEL_ROUTE_DEL, struct gr_mpls_label_route_del_req, struct gr_empty); + +// Get a single label route by label value. +struct gr_mpls_label_route_get_req { + uint16_t vrf_id; + uint32_t in_label; +}; + +GR_REQ(GR_MPLS_LABEL_ROUTE_GET, struct gr_mpls_label_route_get_req, struct gr_mpls_label_route); + +// List all label routes in a VRF. +struct gr_mpls_label_route_list_req { + uint16_t vrf_id; + uint16_t max_count; +}; + +GR_REQ_STREAM( + GR_MPLS_LABEL_ROUTE_LIST, + struct gr_mpls_label_route_list_req, + struct gr_mpls_label_route +); + +// events + +enum gr_mpls_events : uint32_t { + GR_EVENT_MPLS_ROUTE_ADD = GR_MSG_TYPE(GR_MPLS_MODULE, 0x1001), + GR_EVENT_MPLS_ROUTE_DEL, +}; + +GR_EVENT(GR_EVENT_MPLS_ROUTE_ADD, struct gr_mpls_label_route); +GR_EVENT(GR_EVENT_MPLS_ROUTE_DEL, struct gr_mpls_label_route); From 8e01a8e97e7a7f87c4ef88cbf76ace46e426ff20 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Mon, 22 Jun 2026 15:29:57 +0200 Subject: [PATCH 04/11] mpls: add control plane module MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Implement the MPLS nexthop type operations, the per-VRF label forwarding information base, and the API request handlers. The nexthop type ops handle creation, update, and teardown of MPLS nexthops. Each MPLS nexthop holds a reference to an L3 nexthop for gateway resolution. The LFIB is a flat array of nexthop pointers indexed directly by the 20-bit label value, allocated per VRF via the vrf_fib_ops mechanism. Signed-off-by: Matej Mužila --- modules/meson.build | 1 + modules/mpls/api/gr_mpls.h | 44 +++++++ modules/mpls/api/meson.build | 5 + modules/mpls/control/label.c | 192 ++++++++++++++++++++++++++++ modules/mpls/control/label_table.c | 199 +++++++++++++++++++++++++++++ modules/mpls/control/meson.build | 8 ++ modules/mpls/control/mpls.h | 26 ++++ modules/mpls/control/nexthop.c | 143 +++++++++++++++++++++ modules/mpls/meson.build | 5 + 9 files changed, 623 insertions(+) create mode 100644 modules/mpls/api/meson.build create mode 100644 modules/mpls/control/label.c create mode 100644 modules/mpls/control/label_table.c create mode 100644 modules/mpls/control/meson.build create mode 100644 modules/mpls/control/mpls.h create mode 100644 modules/mpls/control/nexthop.c create mode 100644 modules/mpls/meson.build diff --git a/modules/meson.build b/modules/meson.build index 0b978a62b..8ddb1b3a2 100644 --- a/modules/meson.build +++ b/modules/meson.build @@ -5,6 +5,7 @@ subdir('infra') subdir('ip') subdir('ip6') subdir('ipip') +subdir('mpls') subdir('l2') subdir('l4') subdir('policy') diff --git a/modules/mpls/api/gr_mpls.h b/modules/mpls/api/gr_mpls.h index a5b590516..020f061b5 100644 --- a/modules/mpls/api/gr_mpls.h +++ b/modules/mpls/api/gr_mpls.h @@ -99,3 +99,47 @@ enum gr_mpls_events : uint32_t { GR_EVENT(GR_EVENT_MPLS_ROUTE_ADD, struct gr_mpls_label_route); GR_EVENT(GR_EVENT_MPLS_ROUTE_DEL, struct gr_mpls_label_route); + +// label range reservation + +enum gr_mpls_label_range_requests : uint32_t { + GR_MPLS_LABEL_RANGE_ADD = GR_MSG_TYPE(GR_MPLS_MODULE, 0x0010), + GR_MPLS_LABEL_RANGE_DEL, + GR_MPLS_LABEL_RANGE_LIST, +}; + +struct gr_mpls_label_range { + uint16_t vrf_id; + uint32_t start; + uint32_t end; + gr_nh_origin_t origin; +}; + +struct gr_mpls_label_range_add_req { + uint16_t vrf_id; + uint32_t start; + uint32_t end; + gr_nh_origin_t origin; + uint8_t exist_ok; +}; + +GR_REQ(GR_MPLS_LABEL_RANGE_ADD, struct gr_mpls_label_range_add_req, struct gr_empty); + +struct gr_mpls_label_range_del_req { + uint16_t vrf_id; + uint32_t start; + uint32_t end; + uint8_t missing_ok; +}; + +GR_REQ(GR_MPLS_LABEL_RANGE_DEL, struct gr_mpls_label_range_del_req, struct gr_empty); + +struct gr_mpls_label_range_list_req { + uint16_t vrf_id; +}; + +GR_REQ_STREAM( + GR_MPLS_LABEL_RANGE_LIST, + struct gr_mpls_label_range_list_req, + struct gr_mpls_label_range +); diff --git a/modules/mpls/api/meson.build b/modules/mpls/api/meson.build new file mode 100644 index 000000000..c15b0d016 --- /dev/null +++ b/modules/mpls/api/meson.build @@ -0,0 +1,5 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +api_headers += files('gr_mpls.h') +api_inc += include_directories('.') diff --git a/modules/mpls/control/label.c b/modules/mpls/control/label.c new file mode 100644 index 000000000..5e74e671e --- /dev/null +++ b/modules/mpls/control/label.c @@ -0,0 +1,192 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "event.h" +#include "module.h" +#include "mpls.h" + +#include + +static struct api_out label_route_add(const void *request, struct api_ctx *) { + const struct gr_mpls_label_route_add_req *req = request; + struct nexthop *nh; + int ret; + + if (req->in_label > GR_MPLS_LABEL_MAX) + return api_out(EINVAL, 0, NULL); + + nh = nexthop_lookup_id(req->nh_id); + if (nh == NULL) + return api_out(ENOENT, 0, NULL); + if (nh->type != GR_NH_T_MPLS && nh->type != GR_NH_T_GROUP) + return api_out(EINVAL, 0, NULL); + + ret = mpls_rib_insert(req->vrf_id, req->in_label, nh, req->origin, req->exist_ok); + + return api_out(-ret, 0, NULL); +} + +static struct api_out label_route_del(const void *request, struct api_ctx *) { + const struct gr_mpls_label_route_del_req *req = request; + int ret; + + ret = mpls_rib_delete(req->vrf_id, req->in_label, req->missing_ok); + + return api_out(-ret, 0, NULL); +} + +static struct api_out label_route_get(const void *request, struct api_ctx *) { + const struct gr_mpls_label_route_get_req *req = request; + struct gr_mpls_label_route *resp; + const struct nexthop *nh; + + nh = mpls_fib_lookup(req->vrf_id, req->in_label); + if (nh == NULL) + return api_out(ENOENT, 0, NULL); + + resp = calloc(1, sizeof(*resp)); + if (resp == NULL) + return api_out(ENOMEM, 0, NULL); + + resp->vrf_id = req->vrf_id; + resp->in_label = req->in_label; + resp->nh_id = nh->nh_id; + resp->origin = nh->origin; + + return api_out(0, sizeof(*resp), resp); +} + +struct label_list_ctx { + struct api_ctx *ctx; + uint16_t max_count; + uint16_t count; +}; + +static int label_route_send(uint16_t vrf_id, uint32_t label, const struct nexthop *nh, void *priv) { + struct label_list_ctx *lctx = priv; + struct gr_mpls_label_route route; + + if (lctx->max_count != 0 && lctx->count >= lctx->max_count) + return errno_set(EXFULL); + + route = (struct gr_mpls_label_route) { + .vrf_id = vrf_id, + .in_label = label, + .nh_id = nh->nh_id, + .origin = nh->origin, + }; + + api_send(lctx->ctx, sizeof(route), &route); + lctx->count++; + + return 0; +} + +static struct api_out label_route_list(const void *request, struct api_ctx *ctx) { + const struct gr_mpls_label_route_list_req *req = request; + struct label_list_ctx lctx; + int ret; + + lctx = (struct label_list_ctx) { + .ctx = ctx, + .max_count = req->max_count, + }; + + ret = mpls_rib_iter(req->vrf_id, label_route_send, &lctx); + if (ret == -EXFULL) + ret = 0; + + return api_out(-ret, 0, NULL); +} + +// label range reservation — simple linked list; small number of entries expected + +struct label_range_entry { + struct gr_mpls_label_range range; + struct label_range_entry *next; +}; + +static struct label_range_entry *label_ranges; + +static struct api_out label_range_add(const void *request, struct api_ctx *) { + const struct gr_mpls_label_range_add_req *req = request; + struct label_range_entry *e; + + if (req->start > req->end || req->end > GR_MPLS_LABEL_MAX) + return api_out(EINVAL, 0, NULL); + + for (struct label_range_entry *ex = label_ranges; ex != NULL; ex = ex->next) { + if (ex->range.vrf_id != req->vrf_id) + continue; + if (ex->range.start == req->start && ex->range.end == req->end) { + if (!req->exist_ok) + return api_out(EEXIST, 0, NULL); + return api_out(0, 0, NULL); + } + } + + e = calloc(1, sizeof(*e)); + if (e == NULL) + return api_out(ENOMEM, 0, NULL); + + e->range.vrf_id = req->vrf_id; + e->range.start = req->start; + e->range.end = req->end; + e->range.origin = req->origin; + e->next = label_ranges; + label_ranges = e; + + return api_out(0, 0, NULL); +} + +static struct api_out label_range_del(const void *request, struct api_ctx *) { + const struct gr_mpls_label_range_del_req *req = request; + struct label_range_entry **pp; + + pp = &label_ranges; + + while (*pp != NULL) { + struct label_range_entry *e = *pp; + if (e->range.vrf_id == req->vrf_id && e->range.start == req->start + && e->range.end == req->end) { + *pp = e->next; + free(e); + return api_out(0, 0, NULL); + } + pp = &e->next; + } + + if (!req->missing_ok) + return api_out(ENOENT, 0, NULL); + return api_out(0, 0, NULL); +} + +static struct api_out label_range_list(const void *request, struct api_ctx *ctx) { + const struct gr_mpls_label_range_list_req *req = request; + + for (struct label_range_entry *e = label_ranges; e != NULL; e = e->next) { + if (req->vrf_id != 0 && e->range.vrf_id != req->vrf_id) + continue; + api_send(ctx, sizeof(e->range), &e->range); + } + + return api_out(0, 0, NULL); +} + +static struct module mpls_module = { + .name = "mpls", + .depends_on = "nexthop", +}; + +RTE_INIT(mpls_constructor) { + module_register(&mpls_module); + api_handler(GR_MPLS_LABEL_ROUTE_ADD, label_route_add); + api_handler(GR_MPLS_LABEL_ROUTE_DEL, label_route_del); + api_handler(GR_MPLS_LABEL_ROUTE_GET, label_route_get); + api_handler(GR_MPLS_LABEL_ROUTE_LIST, label_route_list); + event_serializer(GR_EVENT_MPLS_ROUTE_ADD, NULL); + event_serializer(GR_EVENT_MPLS_ROUTE_DEL, NULL); + api_handler(GR_MPLS_LABEL_RANGE_ADD, label_range_add); + api_handler(GR_MPLS_LABEL_RANGE_DEL, label_range_del); + api_handler(GR_MPLS_LABEL_RANGE_LIST, label_range_list); +} diff --git a/modules/mpls/control/label_table.c b/modules/mpls/control/label_table.c new file mode 100644 index 000000000..3984326f6 --- /dev/null +++ b/modules/mpls/control/label_table.c @@ -0,0 +1,199 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "config.h" +#include "event.h" +#include "log.h" +#include "mpls.h" + +#include + +#include + +LOG_TYPE("mpls"); + +#define MPLS_LFIB_SIZE (GR_MPLS_LABEL_MAX + 1) + +static struct nexthop **get_lfib(uint16_t vrf_id) { + struct nexthop **lfib; + struct iface *iface; + + iface = get_vrf_iface(vrf_id); + if (iface == NULL) + return NULL; + lfib = iface_info_vrf(iface)->fib_mpls; + if (lfib == NULL) + return errno_set_null(ENONET); + return lfib; +} + +const struct nexthop *mpls_fib_lookup(uint16_t vrf_id, uint32_t label) { + struct nexthop **lfib; + + lfib = get_lfib(vrf_id); + if (lfib == NULL || label > GR_MPLS_LABEL_MAX) + return NULL; + + return lfib[label]; +} + +int mpls_rib_insert( + uint16_t vrf_id, + uint32_t label, + struct nexthop *nh, + gr_nh_origin_t origin, + bool exist_ok +) { + struct nexthop *existing; + struct nexthop **lfib; + + lfib = get_lfib(vrf_id); + + if (lfib == NULL) + return -errno; + if (label > GR_MPLS_LABEL_MAX) + return errno_set(EINVAL); + + existing = lfib[label]; + if (existing != NULL) { + if (!exist_ok) + return errno_set(EEXIST); + if (existing == nh) + return 0; + } + + nexthop_incref(nh); + lfib[label] = nh; + + if (origin != GR_NH_ORIGIN_INTERNAL) { + event_push( + GR_EVENT_MPLS_ROUTE_ADD, + &(const struct gr_mpls_label_route) { + .vrf_id = vrf_id, + .in_label = label, + .nh_id = nh->nh_id, + .origin = origin, + } + ); + } + + if (existing != NULL) + nexthop_decref(existing); + + return 0; +} + +int mpls_rib_delete(uint16_t vrf_id, uint32_t label, bool missing_ok) { + struct nexthop **lfib; + struct nexthop *nh; + + lfib = get_lfib(vrf_id); + + if (lfib == NULL) + return -errno; + if (label > GR_MPLS_LABEL_MAX) + return errno_set(EINVAL); + + nh = lfib[label]; + if (nh == NULL) { + if (missing_ok) + return 0; + return errno_set(ENOENT); + } + + lfib[label] = NULL; + + if (nh->origin != GR_NH_ORIGIN_INTERNAL) { + event_push( + GR_EVENT_MPLS_ROUTE_DEL, + &(const struct gr_mpls_label_route) { + .vrf_id = vrf_id, + .in_label = label, + .nh_id = nh->nh_id, + .origin = nh->origin, + } + ); + } + + nexthop_decref(nh); + + return 0; +} + +int mpls_rib_iter(uint16_t vrf_id, mpls_rib_iter_cb cb, void *priv) { + struct nexthop **lfib; + struct iface *iface; + int ret; + + if (vrf_id != GR_VRF_ID_UNDEF) { + lfib = get_lfib(vrf_id); + if (lfib == NULL) + return -errno; + for (uint32_t i = 0; i < MPLS_LFIB_SIZE; i++) { + if (lfib[i] == NULL) + continue; + ret = cb(vrf_id, i, lfib[i], priv); + if (ret < 0) + return ret; + } + } else { + for (uint16_t v = 1; v < gr_config.max_ifaces; v++) { + iface = iface_from_id(v); + if (iface == NULL || iface->type != GR_IFACE_TYPE_VRF) + continue; + lfib = iface_info_vrf(iface)->fib_mpls; + if (lfib == NULL) + continue; + for (uint32_t i = 0; i < MPLS_LFIB_SIZE; i++) { + if (lfib[i] == NULL) + continue; + ret = cb(v, i, lfib[i], priv); + if (ret < 0) + return ret; + } + } + } + return 0; +} + +static int mpls_fib_init(struct iface *vrf) { + struct nexthop **lfib; + + lfib = rte_zmalloc("mpls_lfib", MPLS_LFIB_SIZE * sizeof(*lfib), RTE_CACHE_LINE_SIZE); + if (lfib == NULL) + return errno_log(rte_errno, "rte_zmalloc(mpls_lfib)"); + + iface_info_vrf(vrf)->fib_mpls = lfib; + + return 0; +} + +static int mpls_fib_reconfig(struct iface *) { + return 0; +} + +static void mpls_fib_fini(struct iface *vrf) { + struct nexthop **lfib; + + lfib = iface_info_vrf(vrf)->fib_mpls; + if (lfib == NULL) + return; + + for (uint32_t i = 0; i < MPLS_LFIB_SIZE; i++) { + if (lfib[i] != NULL) + nexthop_decref(lfib[i]); + } + + rte_free(lfib); + iface_info_vrf(vrf)->fib_mpls = NULL; +} + +static const struct vrf_fib_ops mpls_fib_ops = { + .init = mpls_fib_init, + .reconfig = mpls_fib_reconfig, + .fini = mpls_fib_fini, +}; + +RTE_INIT(mpls_label_table_init) { + vrf_fib_ops_register(GR_AF_MPLS, &mpls_fib_ops); +} diff --git a/modules/mpls/control/meson.build b/modules/mpls/control/meson.build new file mode 100644 index 000000000..cb726c454 --- /dev/null +++ b/modules/mpls/control/meson.build @@ -0,0 +1,8 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +src += files( + 'label.c', + 'label_table.c', + 'nexthop.c', +) diff --git a/modules/mpls/control/mpls.h b/modules/mpls/control/mpls.h new file mode 100644 index 000000000..bd1fd8898 --- /dev/null +++ b/modules/mpls/control/mpls.h @@ -0,0 +1,26 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#pragma once + +#include "nexthop.h" +#include "vrf.h" + +#include + +GR_NH_TYPE_INFO(GR_NH_T_MPLS, nexthop_info_mpls, { + uint8_t n_labels; + uint8_t ttl; + addr_family_t payload_af; + uint32_t labels[GR_MPLS_MAX_LABELS]; + struct nexthop *via_nh; +}); + +// Look up a label in the per-VRF LFIB. Called from the datapath. +const struct nexthop *mpls_fib_lookup(uint16_t vrf_id, uint32_t label); + +int mpls_rib_insert(uint16_t, uint32_t, struct nexthop *, gr_nh_origin_t, bool exist_ok); +int mpls_rib_delete(uint16_t vrf_id, uint32_t label, bool missing_ok); + +typedef int (*mpls_rib_iter_cb)(uint16_t, uint32_t, const struct nexthop *, void *); +int mpls_rib_iter(uint16_t vrf_id, mpls_rib_iter_cb cb, void *priv); diff --git a/modules/mpls/control/nexthop.c b/modules/mpls/control/nexthop.c new file mode 100644 index 000000000..f91124f90 --- /dev/null +++ b/modules/mpls/control/nexthop.c @@ -0,0 +1,143 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "mpls.h" + +#include + +static int mpls_nh_import_info(struct nexthop *nh, const void *info) { + struct nexthop_info_mpls *priv = nexthop_info_mpls(nh); + const struct gr_nexthop_info_mpls *pub = info; + struct gr_nexthop_info_l3 l3_info; + struct nexthop *via_nh, *old_via; + + if (pub->n_labels > GR_MPLS_MAX_LABELS) + return errno_set(EINVAL); + + for (uint8_t i = 0; i < pub->n_labels; i++) { + if (pub->labels[i] > GR_MPLS_LABEL_MAX) + return errno_set(EINVAL); + if (pub->labels[i] == GR_MPLS_LABEL_IMPLICIT_NULL) + return errno_set(EINVAL); + } + + switch (pub->payload_af) { + case GR_AF_UNSPEC: + case GR_AF_IP4: + case GR_AF_IP6: + break; + default: + return errno_set(EINVAL); + } + + if (pub->via.af != GR_AF_IP4 && pub->via.af != GR_AF_IP6) + return errno_set(EAFNOSUPPORT); + + via_nh = nexthop_lookup_l3(pub->via.af, nh->vrf_id, nh->iface_id, &pub->via.addr); + if (via_nh == NULL) { + l3_info = (struct gr_nexthop_info_l3) {.af = pub->via.af}; + if (pub->via.af == GR_AF_IP4) + l3_info.ipv4 = pub->via.ipv4; + else + l3_info.ipv6 = pub->via.ipv6; + + via_nh = nexthop_new( + &(struct gr_nexthop_base) { + .type = GR_NH_T_L3, + .iface_id = nh->iface_id, + .vrf_id = nh->vrf_id, + .origin = GR_NH_ORIGIN_INTERNAL, + }, + &l3_info + ); + if (via_nh == NULL) + return -errno; + } else { + nexthop_incref(via_nh); + } + + priv->n_labels = pub->n_labels; + priv->ttl = pub->ttl; + priv->payload_af = pub->payload_af; + memcpy(priv->labels, pub->labels, pub->n_labels * sizeof(pub->labels[0])); + + old_via = priv->via_nh; + priv->via_nh = via_nh; + if (old_via != NULL) + nexthop_decref(old_via); + + return 0; +} + +static void mpls_nh_free(struct nexthop *nh) { + struct nexthop_info_mpls *priv = nexthop_info_mpls(nh); + if (priv->via_nh != NULL) + nexthop_decref(priv->via_nh); + priv->via_nh = NULL; +} + +static void mpls_nh_remove_via_cb(struct nexthop *nh, void *dying) { + if (nh->type != GR_NH_T_MPLS) + return; + struct nexthop_info_mpls *priv = nexthop_info_mpls(nh); + if (priv->via_nh == dying) + priv->via_nh = NULL; +} + +static void mpls_nh_remove_references(struct nexthop *dying) { + nexthop_iter(mpls_nh_remove_via_cb, dying); +} + +static bool mpls_nh_equal(const struct nexthop *a, const struct nexthop *b) { + const struct nexthop_info_mpls *ma = nexthop_info_mpls(a); + const struct nexthop_info_mpls *mb = nexthop_info_mpls(b); + + if (ma->n_labels != mb->n_labels) + return false; + if (ma->payload_af != mb->payload_af) + return false; + if (ma->via_nh != mb->via_nh) + return false; + return memcmp(ma->labels, mb->labels, ma->n_labels * sizeof(ma->labels[0])) == 0; +} + +static struct gr_nexthop *mpls_nh_to_api(const struct nexthop *nh, size_t *len) { + const struct nexthop_info_mpls *priv = nexthop_info_mpls(nh); + struct gr_nexthop_info_mpls *pub_info; + struct gr_nexthop *pub; + + *len = sizeof(*pub) + sizeof(*pub_info); + pub = calloc(1, *len); + if (pub == NULL) + return errno_set_null(ENOMEM); + + pub->base = nh->base; + pub_info = (struct gr_nexthop_info_mpls *)pub->info; + pub_info->n_labels = priv->n_labels; + pub_info->ttl = priv->ttl; + pub_info->payload_af = priv->payload_af; + memcpy(pub_info->labels, priv->labels, priv->n_labels * sizeof(priv->labels[0])); + + if (priv->via_nh != NULL) { + const struct nexthop_info_l3 *l3 = nexthop_info_l3(priv->via_nh); + pub_info->via.af = l3->af; + if (l3->af == GR_AF_IP4) + pub_info->via.ipv4 = l3->ipv4; + else if (l3->af == GR_AF_IP6) + pub_info->via.ipv6 = l3->ipv6; + } + + return pub; +} + +static struct nexthop_type_ops mpls_nh_ops = { + .import_info = mpls_nh_import_info, + .free = mpls_nh_free, + .remove_references = mpls_nh_remove_references, + .equal = mpls_nh_equal, + .to_api = mpls_nh_to_api, +}; + +RTE_INIT(mpls_nexthop_init) { + nexthop_type_ops_register(GR_NH_T_MPLS, &mpls_nh_ops); +} diff --git a/modules/mpls/meson.build b/modules/mpls/meson.build new file mode 100644 index 000000000..7d9a19856 --- /dev/null +++ b/modules/mpls/meson.build @@ -0,0 +1,5 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +subdir('api') +subdir('control') From e877dbc7906433e9afb6433d5f5c3d62cf531a93 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Mon, 22 Jun 2026 16:17:01 +0200 Subject: [PATCH 05/11] mpls: add datapath nodes and resolve callbacks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three new graph nodes handle MPLS forwarding: input decodes the label stack in a bounded loop (rte_graph forbids self-loops), dispatching to swap, pop or reserved-label handling with TTL propagation and payload type detection; output resolves the egress MAC and parks unresolved packets in ip_hold via the mbuf priv data overlay so the MPLS nexthop survives ARP resolution; push prepends the label stack from ip_output and ip6_output and accounts for the label overhead in MTU checks. mpls_error handles oversized packets by peeking past the label stack to determine the inner IP version, then generating ICMP frag-needed or ICMPv6 packet-too-big accordingly. Signed-off-by: Matej Mužila --- docs/graph.svg | 1022 ++++++++++++++----------- modules/mpls/control/meson.build | 2 + modules/mpls/control/resolve.c | 98 +++ modules/mpls/datapath/meson.build | 10 + modules/mpls/datapath/mpls_datapath.h | 26 + modules/mpls/datapath/mpls_error.c | 425 ++++++++++ modules/mpls/datapath/mpls_input.c | 251 ++++++ modules/mpls/datapath/mpls_output.c | 167 ++++ modules/mpls/datapath/mpls_push.c | 137 ++++ modules/mpls/meson.build | 1 + 10 files changed, 1700 insertions(+), 439 deletions(-) create mode 100644 modules/mpls/control/resolve.c create mode 100644 modules/mpls/datapath/meson.build create mode 100644 modules/mpls/datapath/mpls_datapath.h create mode 100644 modules/mpls/datapath/mpls_error.c create mode 100644 modules/mpls/datapath/mpls_input.c create mode 100644 modules/mpls/datapath/mpls_output.c create mode 100644 modules/mpls/datapath/mpls_push.c diff --git a/docs/graph.svg b/docs/graph.svg index a3e455385..6051a923d 100644 --- a/docs/graph.svg +++ b/docs/graph.svg @@ -4,957 +4,1101 @@ - - - + + + bond_output - -bond_output + +bond_output port_output - -port_output + +port_output bond_output->port_output - - + + iface_input - -iface_input + +iface_input xconnect - -xconnect + +xconnect - + iface_input->xconnect - - + + eth_input - -eth_input + +eth_input - + iface_input->eth_input - - + + bridge_input - -bridge_input + +bridge_input - + iface_input->bridge_input - - + + iface_output - -iface_output + +iface_output - + iface_output->bond_output - - + + - + iface_output->port_output - - + + - + iface_output->bridge_input - - + + - + vxlan_output - -vxlan_output + +vxlan_output - + iface_output->vxlan_output - - + + port_tx - -port_tx + +port_tx - + port_output->port_tx - - + + port_rx - -port_rx + +port_rx - + port_rx->iface_input - - + + - + xconnect->port_output - - + + lacp_input - -lacp_input + +lacp_input eth_input->lacp_input - - + + snap_input - -snap_input + +snap_input eth_input->snap_input - - + + arp_input - -arp_input + +arp_input eth_input->arp_input - - + + ip_input - -ip_input + +ip_input eth_input->ip_input - - + + ip6_input - -ip6_input + +ip6_input eth_input->ip6_input - - + + + + + +mpls_input + +mpls_input + + + +eth_input->mpls_input + + eth_output - -eth_output + +eth_output - + eth_output->iface_output - - + + l2_redirect - -l2_redirect + +l2_redirect lacp_output - -lacp_output + +lacp_output - + lacp_output->eth_output - - + + - + snap_input->l2_redirect - - + + arp_input_reply - -arp_input_reply + +arp_input_reply - + arp_input->arp_input_reply - - + + arp_input_request - -arp_input_request + +arp_input_request - + arp_input->arp_input_request - - + + arp_output_reply - -arp_output_reply + +arp_output_reply - + arp_output_reply->eth_output - - + + arp_output_request - -arp_output_request + +arp_output_request - + arp_output_request->eth_output - - + + bridge_flood - -bridge_flood + +bridge_flood - + bridge_flood->iface_input - - + + - + bridge_flood->iface_output - - + + vxlan_flood - -vxlan_flood + +vxlan_flood - + bridge_flood->vxlan_flood - - + + - + bridge_input->iface_input - - + + - + bridge_input->iface_output - - + + - + bridge_input->bridge_flood - - + + bridge_neigh_suppress - -bridge_neigh_suppress + +bridge_neigh_suppress - + bridge_input->bridge_neigh_suppress - - + + - + bridge_neigh_suppress->iface_output - - + + - + bridge_neigh_suppress->bridge_flood - - + + - + vxlan_flood->iface_output - - + + ospf_redirect - -ospf_redirect + +ospf_redirect - + ospf_redirect->l2_redirect - - + + loopback_input - -loopback_input + +loopback_input - + loopback_input->ip_input - - + + - + loopback_input->ip6_input - - + + loopback_output - -loopback_output + +loopback_output xvrf - -xvrf + +xvrf - + xvrf->ip_input - - + + - + xvrf->ip6_input - - + + ip_forward - -ip_forward + +ip_forward ip_output - -ip_output + +ip_output - + ip_forward->ip_output - - + + ip_fragment - -ip_fragment + +ip_fragment - + ip_fragment->ip_output - - + + ip_hold - -ip_hold + +ip_hold - + ip_input->ip_forward - - + + ip_input_local - -ip_input_local + +ip_input_local - + ip_input->ip_input_local - - + + - + ip_input->ip_output - - + + - + dnat44_dynamic - -dnat44_dynamic + +dnat44_dynamic - + ip_input->dnat44_dynamic - - + + - + dnat44_static - -dnat44_static + +dnat44_static - + ip_input->dnat44_static - - + + - + ip_input_local->ospf_redirect - - + + ipip_input - -ipip_input + +ipip_input - + ip_input_local->ipip_input - - + + - + icmp_input - -icmp_input + +icmp_input - + ip_input_local->icmp_input - - + + - + l4_input_local - -l4_input_local + +l4_input_local - + ip_input_local->l4_input_local - - + + - + ip_output->eth_output - - + + - + ip_output->xvrf - - + + - + ip_output->ip_fragment - - + + - + ip_output->ip_hold - - + + ipip_output - -ipip_output + +ipip_output - + ip_output->ipip_output - - + + + + + +mpls_push + +mpls_push + + + +ip_output->mpls_push + + - + sr6_output - -sr6_output + +sr6_output - + ip_output->sr6_output - - + + ip6_forward - -ip6_forward + +ip6_forward ip6_output - -ip6_output + +ip6_output - + ip6_forward->ip6_output - - + + ip6_hold - -ip6_hold + +ip6_hold - + ip6_input->ip6_forward - - + + ip6_input_local - -ip6_input_local + +ip6_input_local - + ip6_input->ip6_input_local - - + + - + ip6_input->ip6_output - - + + - + sr6_local - -sr6_local + +sr6_local - + ip6_input->sr6_local - - + + - + ip6_input_local->ospf_redirect - - + + - + icmp6_input - -icmp6_input + +icmp6_input - + ip6_input_local->icmp6_input - - + + - + ip6_input_local->l4_input_local - - + + - + ip6_output->eth_output - - + + - + ip6_output->xvrf - - + + - + ip6_output->ip6_hold - - + + + + + +ip6_output->mpls_push + + - + ip6_output->sr6_output - - + + - + ipip_input->ip_input - - + + - + ipip_output->ip_output - - + + - + +mpls_push_frag_needed + +mpls_push_frag_needed + + + +icmp_output + +icmp_output + + + +mpls_push_frag_needed->icmp_output + + + + + +mpls_push_pkt_too_big + +mpls_push_pkt_too_big + + + +icmp6_output + +icmp6_output + + + +mpls_push_pkt_too_big->icmp6_output + + + + + +mpls_output_frag_needed + +mpls_output_frag_needed + + + +mpls_output_frag_needed->icmp_output + + + + + +mpls_output_pkt_too_big + +mpls_output_pkt_too_big + + + +mpls_output_pkt_too_big->icmp6_output + + + + + +mpls_input->ip_input + + + + + +mpls_input->ip6_input + + + + + +mpls_output + +mpls_output + + + +mpls_input->mpls_output + + + + + +mpls_output->eth_output + + + + + +mpls_output->ip_hold + + + + + +mpls_output->mpls_output_frag_needed + + + + + +mpls_output->mpls_output_pkt_too_big + + + + + +mpls_push->mpls_push_frag_needed + + + + + +mpls_push->mpls_push_pkt_too_big + + + + + +mpls_push->mpls_output + + + + + vxlan_input - -vxlan_input + +vxlan_input - + vxlan_input->iface_input - - + + - + vxlan_output->ip_output - - + + - + vxlan_output->ip6_output - - + + - + dnat44_dynamic->ip_forward - - + + - + dnat44_dynamic->ip_input_local - - + + - + dnat44_static->ip_forward - - + + - + dnat44_static->ip_input_local - - + + - + sr6_local->ip_input - - + + - + sr6_local->ip6_forward - - + + - + sr6_local->ip6_input - - + + - + sr6_local->ip6_input_local - - + + - + sr6_output->ip6_output - - - - - -icmp_output - -icmp_output + + - + icmp_input->icmp_output - - + + - + icmp_local_send - -icmp_local_send + +icmp_local_send - + icmp_local_send->icmp_output - - + + - + icmp_output->ip_output - - - - - -icmp6_output - -icmp6_output + + - + icmp6_input->icmp6_output - - + + - + ndp_na_input - -ndp_na_input + +ndp_na_input - + icmp6_input->ndp_na_input - - + + - + ndp_ns_input - -ndp_ns_input + +ndp_ns_input - + icmp6_input->ndp_ns_input - - + + - + ndp_ra_input - -ndp_ra_input + +ndp_ra_input - + icmp6_input->ndp_ra_input - - + + - + ndp_rs_input - -ndp_rs_input + +ndp_rs_input - + icmp6_input->ndp_rs_input - - + + - + icmp6_local_send - -icmp6_local_send + +icmp6_local_send - + icmp6_local_send->icmp6_output - - + + - + icmp6_output->ip6_output - - + + - + ndp_na_output - -ndp_na_output + +ndp_na_output - + ndp_na_output->icmp6_output - - + + - + ndp_ns_output - -ndp_ns_output + +ndp_ns_output - + ndp_ns_output->icmp6_output - - + + - + l4_loopback_output - -l4_loopback_output + +l4_loopback_output - + ndp_ra_input->l4_loopback_output - - + + - + l4_input_local->vxlan_input - - + + - + l4_input_local->l4_loopback_output - - + + - + dhcp_input - -dhcp_input + +dhcp_input - + l4_input_local->dhcp_input - - + + - + l4_loopback_output->loopback_output - - + + diff --git a/modules/mpls/control/meson.build b/modules/mpls/control/meson.build index cb726c454..b404cc399 100644 --- a/modules/mpls/control/meson.build +++ b/modules/mpls/control/meson.build @@ -5,4 +5,6 @@ src += files( 'label.c', 'label_table.c', 'nexthop.c', + 'resolve.c', ) +inc += include_directories('.') diff --git a/modules/mpls/control/resolve.c b/modules/mpls/control/resolve.c new file mode 100644 index 000000000..3fd5da996 --- /dev/null +++ b/modules/mpls/control/resolve.c @@ -0,0 +1,98 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "config.h" +#include "l3.h" +#include "log.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +LOG_TYPE("mpls"); + +static void mpls_resolve_cb(void *obj, uintptr_t, const struct control_queue_drain *drain) { + struct nexthop_info_l3 *l3; + struct rte_mbuf *m = obj; + struct nexthop *nh; + + nh = (struct nexthop *)l3_mbuf_data(m)->nh; + + if (drain != NULL) { + switch (drain->event) { + case GR_EVENT_IFACE_REMOVE: + if (mbuf_data(m)->iface == drain->obj) + goto free; + break; + case GR_EVENT_NEXTHOP_DELETE: + if (nh == drain->obj) + goto free; + if (mpls_hold_mbuf_data(m)->mpls_nh == drain->obj) + goto free; + break; + } + } + + l3 = nexthop_info_l3(nh); + + if (l3->state == GR_NH_S_REACHABLE) { + if (mpls_resubmit_cb(m, nh) < 0) + goto free; + return; + } + + if (l3->held_pkts < nh_conf.max_held_pkts) { + queue_mbuf_data(m)->next = NULL; + if (l3->held_pkts_head == NULL) + l3->held_pkts_head = m; + else + queue_mbuf_data(l3->held_pkts_tail)->next = m; + l3->held_pkts_tail = m; + l3->held_pkts++; + if (l3->state != GR_NH_S_PENDING) { + const struct nexthop_af_ops *ops = nexthop_af_ops_from_nh(nh); + if (ops != NULL) + ops->solicit(nh); + l3->state = GR_NH_S_PENDING; + } + return; + } + +free: + rte_pktmbuf_free(m); +} + +static int mpls_solicit(struct nexthop *nh) { + const struct nexthop_af_ops *ops = nexthop_af_ops_from_nh(nh); + if (ops != NULL) + return ops->solicit(nh); + return errno_set(ENOTSUP); +} + +static void mpls_rib_cleanup(struct nexthop *nh, bool) { + for (uint16_t v = 1; v < gr_config.max_ifaces; v++) { + struct iface *iface = iface_from_id(v); + if (iface == NULL || iface->type != GR_IFACE_TYPE_VRF) + continue; + struct nexthop **lfib = iface_info_vrf(iface)->fib_mpls; + if (lfib == NULL) + continue; + for (uint32_t i = 0; i <= GR_MPLS_LABEL_MAX; i++) { + if (lfib[i] == nh) { + lfib[i] = NULL; + nexthop_decref(nh); + } + } + } +} + +static struct nexthop_af_ops mpls_af_ops = { + .resolve = mpls_resolve_cb, + .solicit = mpls_solicit, + .resubmit = mpls_resubmit_cb, + .cleanup_routes = mpls_rib_cleanup, +}; + +RTE_INIT(mpls_resolve_constructor) { + nexthop_af_ops_register(GR_AF_MPLS, &mpls_af_ops); +} diff --git a/modules/mpls/datapath/meson.build b/modules/mpls/datapath/meson.build new file mode 100644 index 000000000..c946118f6 --- /dev/null +++ b/modules/mpls/datapath/meson.build @@ -0,0 +1,10 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +src += files( + 'mpls_error.c', + 'mpls_input.c', + 'mpls_output.c', + 'mpls_push.c', +) +inc += include_directories('.') diff --git a/modules/mpls/datapath/mpls_datapath.h b/modules/mpls/datapath/mpls_datapath.h new file mode 100644 index 000000000..aa99bb94f --- /dev/null +++ b/modules/mpls/datapath/mpls_datapath.h @@ -0,0 +1,26 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#pragma once + +#include "mbuf.h" +#include "nexthop.h" + +#include +#include + +static inline uint32_t mpls_hdr_get_label(const struct rte_mpls_hdr *h) { + return (rte_be_to_cpu_16(h->tag_msb) << 4) | h->tag_lsb; +} + +static inline void mpls_hdr_set_label(struct rte_mpls_hdr *h, uint32_t label) { + h->tag_msb = rte_cpu_to_be_16(label >> 4); + h->tag_lsb = label & 0xf; +} + +GR_MBUF_PRIV_DATA_TYPE(mpls_hold_mbuf_data, { + const struct nexthop *nh; + const struct nexthop *mpls_nh; +}); + +int mpls_resubmit_cb(struct rte_mbuf *, struct nexthop *); diff --git a/modules/mpls/datapath/mpls_error.c b/modules/mpls/datapath/mpls_error.c new file mode 100644 index 000000000..2ca0aa6ac --- /dev/null +++ b/modules/mpls/datapath/mpls_error.c @@ -0,0 +1,425 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "graph.h" +#include "icmp6.h" +#include "ip4.h" +#include "ip4_datapath.h" +#include "ip6.h" +#include "ip6_datapath.h" +#include "l3.h" +#include "mbuf.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include +#include +#include +#include +#include + +#include + +enum { + ICMP_OUTPUT = 0, + NO_HEADROOM, + NO_IP, + EDGE_COUNT, +}; + +static uint16_t mpls_effective_mtu(struct rte_mbuf *mbuf) { + const struct nexthop *nh = l3_mbuf_data(mbuf)->nh; + const struct nexthop_info_mpls *info = nexthop_info_mpls(nh); + const struct iface *out_iface = iface_from_id(info->via_nh->iface_id); + if (out_iface == NULL) + return 0; + return out_iface->mtu - info->n_labels * sizeof(struct rte_mpls_hdr); +} + +static uint16_t mpls_frag_needed_process( + struct rte_graph *graph, + struct rte_node *node, + void **objs, + uint16_t nb_objs +) { + struct ip_local_mbuf_data *ip_data; + const struct nexthop_info_l3 *l3; + const struct nexthop *nh, *local; + const struct iface *in_iface; + struct rte_icmp_hdr *icmp; + struct rte_ipv4_hdr *ip; + uint16_t effective_mtu; + struct rte_mbuf *mbuf; + ip4_addr_t src, dst; + rte_edge_t edge; + unsigned len; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + effective_mtu = mpls_effective_mtu(mbuf); + + ip = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + src = ip->src_addr; + // RFC 792: IP header + 64 bits of original datagram + len = rte_ipv4_hdr_len(ip) + 8; + rte_pktmbuf_trim(mbuf, rte_pktmbuf_pkt_len(mbuf) - len); + + icmp = gr_mbuf_prepend(mbuf, icmp); + if (unlikely(icmp == NULL)) { + edge = NO_HEADROOM; + goto next; + } + + in_iface = mbuf_data(mbuf)->iface; + if (in_iface == NULL || (nh = fib4_lookup(in_iface->vrf_id, src, 0)) == NULL) { + edge = NO_IP; + goto next; + } + if (nh->type == GR_NH_T_L3) { + l3 = nexthop_info_l3(nh); + dst = l3->ipv4; + } else { + dst = src; + } + if ((local = addr4_get_preferred(nh->iface_id, dst)) == NULL) { + edge = NO_IP; + goto next; + } + + icmp->icmp_type = RTE_ICMP_TYPE_DEST_UNREACHABLE; + icmp->icmp_code = RTE_ICMP_CODE_UNREACH_FRAG; + icmp->icmp_cksum = 0; + icmp->icmp_ident = 0; + icmp->icmp_seq_nb = rte_cpu_to_be_16(effective_mtu); + + l3 = nexthop_info_l3(local); + ip_data = ip_local_mbuf_data(mbuf); + ip_data->src = l3->ipv4; + ip_data->dst = src; + ip_data->vrf_id = in_iface->vrf_id; + ip_data->len = rte_pktmbuf_pkt_len(mbuf); + ip_data->proto = IPPROTO_ICMP; + + edge = ICMP_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static uint16_t mpls_pkt_too_big_process( + struct rte_graph *graph, + struct rte_node *node, + void **objs, + uint16_t nb_objs +) { + struct icmp6_err_pkt_too_big *ptb; + const struct nexthop_info_l3 *l3; + struct ip6_local_mbuf_data *d; + const struct iface *in_iface; + const struct nexthop *local; + struct rte_ipv6_hdr *ip6; + uint16_t effective_mtu; + struct rte_mbuf *mbuf; + struct icmp6 *icmp6; + rte_edge_t edge; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + effective_mtu = mpls_effective_mtu(mbuf); + + ip6 = rte_pktmbuf_mtod(mbuf, struct rte_ipv6_hdr *); + + // RFC 4443: as much of the invoking packet as possible without + // the ICMPv6 packet exceeding the minimum IPv6 MTU (1280) + if (rte_pktmbuf_pkt_len(mbuf) > RTE_IPV6_MIN_MTU) + rte_pktmbuf_trim(mbuf, rte_pktmbuf_pkt_len(mbuf) - RTE_IPV6_MIN_MTU); + + ptb = gr_mbuf_prepend(mbuf, ptb); + if (unlikely(ptb == NULL)) { + edge = NO_HEADROOM; + goto next; + } + ptb->mtu = rte_cpu_to_be_32(effective_mtu); + + icmp6 = gr_mbuf_prepend(mbuf, icmp6); + if (unlikely(icmp6 == NULL)) { + edge = NO_HEADROOM; + goto next; + } + icmp6->type = ICMP6_ERR_PKT_TOO_BIG; + icmp6->code = 0; + + in_iface = mbuf_data(mbuf)->iface; + if (in_iface == NULL) { + edge = NO_IP; + goto next; + } + if ((local = addr6_get_preferred(in_iface->id, &ip6->src_addr)) == NULL) { + edge = NO_IP; + goto next; + } + + l3 = nexthop_info_l3(local); + d = ip6_local_mbuf_data(mbuf); + d->src = l3->ipv6; + d->dst = ip6->src_addr; + d->len = rte_pktmbuf_pkt_len(mbuf); + d->iface = in_iface; + + edge = ICMP_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static bool mpls_strip_label_stack(struct rte_mbuf *mbuf) { + struct rte_mpls_hdr *m; + uint32_t depth = 0; + + m = rte_pktmbuf_mtod(mbuf, struct rte_mpls_hdr *); + + while (depth < GR_MPLS_MAX_STACK_DEPTH) { + if (m->bs) { + rte_pktmbuf_adj(mbuf, (depth + 1) * sizeof(*m)); + return true; + } + m++; + depth++; + } + return false; +} + +static uint16_t mpls_output_frag_needed_process( + struct rte_graph *graph, + struct rte_node *node, + void **objs, + uint16_t nb_objs +) { + struct ip_local_mbuf_data *ip_data; + const struct nexthop_info_l3 *l3; + const struct nexthop *nh, *local; + const struct iface *in_iface; + struct rte_icmp_hdr *icmp; + struct rte_ipv4_hdr *ip; + uint16_t effective_mtu; + struct rte_mbuf *mbuf; + ip4_addr_t src, dst; + rte_edge_t edge; + unsigned len; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + effective_mtu = mpls_effective_mtu(mbuf); + + if (!mpls_strip_label_stack(mbuf)) { + edge = NO_IP; + goto next; + } + + ip = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + src = ip->src_addr; + len = rte_ipv4_hdr_len(ip) + 8; + rte_pktmbuf_trim(mbuf, rte_pktmbuf_pkt_len(mbuf) - len); + + icmp = gr_mbuf_prepend(mbuf, icmp); + if (unlikely(icmp == NULL)) { + edge = NO_HEADROOM; + goto next; + } + + in_iface = mbuf_data(mbuf)->iface; + if (in_iface == NULL || (nh = fib4_lookup(in_iface->vrf_id, src, 0)) == NULL) { + edge = NO_IP; + goto next; + } + if (nh->type == GR_NH_T_L3) { + l3 = nexthop_info_l3(nh); + dst = l3->ipv4; + } else { + dst = src; + } + if ((local = addr4_get_preferred(nh->iface_id, dst)) == NULL) { + edge = NO_IP; + goto next; + } + + icmp->icmp_type = RTE_ICMP_TYPE_DEST_UNREACHABLE; + icmp->icmp_code = RTE_ICMP_CODE_UNREACH_FRAG; + icmp->icmp_cksum = 0; + icmp->icmp_ident = 0; + icmp->icmp_seq_nb = rte_cpu_to_be_16(effective_mtu); + + l3 = nexthop_info_l3(local); + ip_data = ip_local_mbuf_data(mbuf); + ip_data->src = l3->ipv4; + ip_data->dst = src; + ip_data->vrf_id = in_iface->vrf_id; + ip_data->len = rte_pktmbuf_pkt_len(mbuf); + ip_data->proto = IPPROTO_ICMP; + + edge = ICMP_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static uint16_t mpls_output_pkt_too_big_process( + struct rte_graph *graph, + struct rte_node *node, + void **objs, + uint16_t nb_objs +) { + struct icmp6_err_pkt_too_big *ptb; + const struct nexthop_info_l3 *l3; + struct ip6_local_mbuf_data *d; + const struct iface *in_iface; + const struct nexthop *local; + struct rte_ipv6_hdr *ip6; + uint16_t effective_mtu; + struct rte_mbuf *mbuf; + struct icmp6 *icmp6; + rte_edge_t edge; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + effective_mtu = mpls_effective_mtu(mbuf); + + if (!mpls_strip_label_stack(mbuf)) { + edge = NO_IP; + goto next; + } + + ip6 = rte_pktmbuf_mtod(mbuf, struct rte_ipv6_hdr *); + + if (rte_pktmbuf_pkt_len(mbuf) > RTE_IPV6_MIN_MTU) + rte_pktmbuf_trim(mbuf, rte_pktmbuf_pkt_len(mbuf) - RTE_IPV6_MIN_MTU); + + ptb = gr_mbuf_prepend(mbuf, ptb); + if (unlikely(ptb == NULL)) { + edge = NO_HEADROOM; + goto next; + } + ptb->mtu = rte_cpu_to_be_32(effective_mtu); + + icmp6 = gr_mbuf_prepend(mbuf, icmp6); + if (unlikely(icmp6 == NULL)) { + edge = NO_HEADROOM; + goto next; + } + icmp6->type = ICMP6_ERR_PKT_TOO_BIG; + icmp6->code = 0; + + in_iface = mbuf_data(mbuf)->iface; + if (in_iface == NULL) { + edge = NO_IP; + goto next; + } + if ((local = addr6_get_preferred(in_iface->id, &ip6->src_addr)) == NULL) { + edge = NO_IP; + goto next; + } + + l3 = nexthop_info_l3(local); + d = ip6_local_mbuf_data(mbuf); + d->src = l3->ipv6; + d->dst = ip6->src_addr; + d->len = rte_pktmbuf_pkt_len(mbuf); + d->iface = in_iface; + + edge = ICMP_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static struct rte_node_register frag_needed_node = { + .name = "mpls_push_frag_needed", + .process = mpls_frag_needed_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, +}; + +static struct rte_node_register pkt_too_big_node = { + .name = "mpls_push_pkt_too_big", + .process = mpls_pkt_too_big_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp6_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, +}; + +static struct gr_node_info frag_needed_info = { + .node = &frag_needed_node, + .type = GR_NODE_T_L3, +}; + +static struct gr_node_info pkt_too_big_info = { + .node = &pkt_too_big_node, + .type = GR_NODE_T_L3, +}; + +GR_NODE_REGISTER(frag_needed_info); +GR_NODE_REGISTER(pkt_too_big_info); + +static struct rte_node_register output_frag_needed_node = { + .name = "mpls_output_frag_needed", + .process = mpls_output_frag_needed_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, +}; + +static struct rte_node_register output_pkt_too_big_node = { + .name = "mpls_output_pkt_too_big", + .process = mpls_output_pkt_too_big_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp6_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, +}; + +static struct gr_node_info output_frag_needed_info = { + .node = &output_frag_needed_node, + .type = GR_NODE_T_L3, +}; + +static struct gr_node_info output_pkt_too_big_info = { + .node = &output_pkt_too_big_node, + .type = GR_NODE_T_L3, +}; + +GR_NODE_REGISTER(output_frag_needed_info); +GR_NODE_REGISTER(output_pkt_too_big_info); diff --git a/modules/mpls/datapath/mpls_input.c b/modules/mpls/datapath/mpls_input.c new file mode 100644 index 000000000..c4fd1aa18 --- /dev/null +++ b/modules/mpls/datapath/mpls_input.c @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "checksum.h" +#include "eth.h" +#include "graph.h" +#include "l3.h" +#include "mbuf.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include +#include +#include + +enum { + MPLS_OUTPUT = 0, + IP_INPUT, + IP6_INPUT, + TTL_EXCEEDED, + NO_ROUTE, + BAD_LABEL, + NO_HEADROOM, + EDGE_COUNT, +}; + +struct trace_mpls_data { + uint32_t label; + uint8_t tc; + uint8_t bs; + uint8_t ttl; +}; + +static int mpls_trace_format(char *buf, size_t len, const void *data, size_t /*data_len*/) { + const struct trace_mpls_data *t = data; + return snprintf(buf, len, "label=%u tc=%u bs=%u ttl=%u", t->label, t->tc, t->bs, t->ttl); +} + +static uint16_t +mpls_input_process(struct rte_graph *graph, struct rte_node *node, void **objs, uint16_t nb_objs) { + const struct nexthop_info_mpls *info; + struct nexthop_info_group *nhg; + struct rte_mpls_hdr trace_hdr; + const struct iface *iface; + struct rte_mpls_hdr *mpls; + const struct nexthop *nh; + struct rte_ipv6_hdr *ip6; + struct rte_ipv4_hdr *ip; + struct rte_mbuf *mbuf; + addr_family_t af; + rte_edge_t edge; + uint32_t label; + uint8_t ver2; + uint8_t bos; + uint8_t ver; + uint8_t ttl; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + edge = BAD_LABEL; + + trace_hdr = *rte_pktmbuf_mtod(mbuf, struct rte_mpls_hdr *); + + for (uint8_t depth = 0; depth < GR_MPLS_MAX_STACK_DEPTH; depth++) { + mpls = rte_pktmbuf_mtod(mbuf, struct rte_mpls_hdr *); + label = mpls_hdr_get_label(mpls); + ttl = mpls->ttl; + bos = mpls->bs; + + if (ttl <= 1) { + edge = TTL_EXCEEDED; + break; + } + ttl -= 1; + + if (label < GR_MPLS_LABEL_FIRST_UNRESERVED) { + rte_pktmbuf_adj(mbuf, sizeof(*mpls)); + switch (label) { + case GR_MPLS_LABEL_IPV4_EXPLICIT_NULL: + mbuf->packet_type = RTE_PTYPE_L3_IPV4; + edge = IP_INPUT; + break; + case GR_MPLS_LABEL_IPV6_EXPLICIT_NULL: + mbuf->packet_type = RTE_PTYPE_L3_IPV6; + edge = IP6_INPUT; + break; + case GR_MPLS_LABEL_IMPLICIT_NULL:; + ver = *rte_pktmbuf_mtod(mbuf, uint8_t *) >> 4; + if (ver == 4) { + mbuf->packet_type = RTE_PTYPE_L3_IPV4; + edge = IP_INPUT; + } else if (ver == 6) { + mbuf->packet_type = RTE_PTYPE_L3_IPV6; + edge = IP6_INPUT; + } else { + edge = BAD_LABEL; + } + break; + case GR_MPLS_LABEL_ROUTER_ALERT: + // RFC 2711: strip label and process payload if BOS, + // otherwise continue to next label in stack. + if (bos) { + ver2 = *rte_pktmbuf_mtod(mbuf, uint8_t *) >> 4; + if (ver2 == 4) { + mbuf->packet_type = RTE_PTYPE_L3_IPV4; + edge = IP_INPUT; + } else if (ver2 == 6) { + mbuf->packet_type = RTE_PTYPE_L3_IPV6; + edge = IP6_INPUT; + } else { + edge = BAD_LABEL; + } + break; + } + continue; + default: + edge = BAD_LABEL; + break; + } + break; + } + + iface = mbuf_data(mbuf)->iface; + nh = mpls_fib_lookup(iface->vrf_id, label); + if (nh == NULL) { + edge = NO_ROUTE; + break; + } + + if (nh->type == GR_NH_T_GROUP) { + nhg = nexthop_info_group(nh); + nh = nexthop_group_get_nh(nhg, mbuf->hash.rss); + if (nh == NULL) { + edge = NO_ROUTE; + break; + } + } + + info = nexthop_info_mpls(nh); + + if (info->n_labels > 0) { + if (info->n_labels > 1) { + mpls = gr_mbuf_prepend( + mbuf, mpls, (info->n_labels - 1) * sizeof(*mpls) + ); + if (unlikely(mpls == NULL)) { + edge = NO_HEADROOM; + break; + } + } + for (uint8_t k = 0; k < info->n_labels; k++) { + mpls_hdr_set_label(&mpls[k], info->labels[k]); + mpls[k].ttl = ttl; + if (k < info->n_labels - 1) { + mpls[k].bs = 0; + mpls[k].tc = 0; + } + } + l3_mbuf_data(mbuf)->nh = nh; + mbuf->packet_type = RTE_PTYPE_TUNNEL_MPLS_IN_GRE; + edge = MPLS_OUTPUT; + break; + } + + if (bos) { + if (info->via_nh == NULL) { + edge = NO_ROUTE; + break; + } + rte_pktmbuf_adj(mbuf, sizeof(*mpls)); + af = info->payload_af; + if (af == GR_AF_UNSPEC) { + ver = *rte_pktmbuf_mtod(mbuf, uint8_t *) >> 4; + if (ver == 4) + af = GR_AF_IP4; + else if (ver == 6) + af = GR_AF_IP6; + } + if (af == GR_AF_IP4) { + ip = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + ip->hdr_checksum = fixup_checksum_16( + ip->hdr_checksum, + rte_cpu_to_be_16(ip->time_to_live << 8), + rte_cpu_to_be_16(ttl << 8) + ); + ip->time_to_live = ttl; + mbuf->packet_type = RTE_PTYPE_L3_IPV4; + } else if (af == GR_AF_IP6) { + ip6 = rte_pktmbuf_mtod(mbuf, struct rte_ipv6_hdr *); + ip6->hop_limits = ttl; + mbuf->packet_type = RTE_PTYPE_L3_IPV6; + } else { + edge = BAD_LABEL; + break; + } + l3_mbuf_data(mbuf)->nh = info->via_nh; + edge = MPLS_OUTPUT; + break; + } + + rte_pktmbuf_adj(mbuf, sizeof(*mpls)); + } + + if (gr_mbuf_is_traced(mbuf)) { + struct trace_mpls_data *t = gr_mbuf_trace_add(mbuf, node, sizeof(*t)); + t->label = mpls_hdr_get_label(&trace_hdr); + t->tc = trace_hdr.tc; + t->bs = trace_hdr.bs; + t->ttl = trace_hdr.ttl; + } + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static void mpls_input_register(void) { + gr_eth_input_add_type(RTE_BE16(RTE_ETHER_TYPE_MPLS), "mpls_input"); +} + +static struct rte_node_register mpls_input_node = { + .name = "mpls_input", + + .process = mpls_input_process, + + .nb_edges = EDGE_COUNT, + .next_nodes = { + [MPLS_OUTPUT] = "mpls_output", + [IP_INPUT] = "ip_input", + [IP6_INPUT] = "ip6_input", + [TTL_EXCEEDED] = "mpls_input_ttl_exceeded", + [NO_ROUTE] = "mpls_input_no_route", + [BAD_LABEL] = "mpls_input_bad_label", + [NO_HEADROOM] = "error_no_headroom", + }, +}; + +static struct gr_node_info info = { + .node = &mpls_input_node, + .type = GR_NODE_T_L3, + .trace_format = mpls_trace_format, + .register_callback = mpls_input_register, +}; + +GR_NODE_REGISTER(info); + +GR_DROP_REGISTER(mpls_input_ttl_exceeded); +GR_DROP_REGISTER(mpls_input_no_route); +GR_DROP_REGISTER(mpls_input_bad_label); diff --git a/modules/mpls/datapath/mpls_output.c b/modules/mpls/datapath/mpls_output.c new file mode 100644 index 000000000..b5bb3dfd9 --- /dev/null +++ b/modules/mpls/datapath/mpls_output.c @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "control_input.h" +#include "eth.h" +#include "graph.h" +#include "l3.h" +#include "log.h" +#include "mbuf.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include +#include +#include +#include +#include + +LOG_TYPE("mpls"); + +static rte_edge_t mpls_output_ctrl; + +int mpls_resubmit_cb(struct rte_mbuf *m, struct nexthop *) { + l3_mbuf_data(m)->nh = mpls_hold_mbuf_data(m)->mpls_nh; + mbuf_data(m)->iface = NULL; + if (post_to_stack(mpls_output_ctrl, m) < 0) { + LOG(ERR, "post_to_stack: %s", strerror(errno)); + return -errno; + } + return 0; +} + +enum { + ETH_OUTPUT = 0, + HOLD, + NO_ROUTE, + TRANSIT_FRAG_NEEDED, + TRANSIT_PKT_TOO_BIG, + MTU_EXCEEDED, + EDGE_COUNT, +}; + +static uint16_t +mpls_output_process(struct rte_graph *graph, struct rte_node *node, void **objs, uint16_t nb_objs) { + struct eth_output_mbuf_data *eth_data; + const struct nexthop_info_l3 *l3; + const struct rte_ipv4_hdr *ip4; + const struct nexthop *via_nh; + const struct iface *iface; + const struct nexthop *nh; + struct rte_mpls_hdr *m; + struct rte_mbuf *mbuf; + rte_edge_t edge; + uint32_t depth; + uint8_t ver; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + nh = l3_mbuf_data(mbuf)->nh; + if (nh == NULL) { + edge = NO_ROUTE; + goto next; + } + + if (nh->type == GR_NH_T_MPLS) { + via_nh = nexthop_info_mpls(nh)->via_nh; + } else if (nh->type == GR_NH_T_L3) { + via_nh = nh; + } else { + edge = NO_ROUTE; + goto next; + } + + if (via_nh == NULL) { + edge = NO_ROUTE; + goto next; + } + + l3 = nexthop_info_l3(via_nh); + iface = iface_from_id(via_nh->iface_id); + if (iface == NULL) { + edge = NO_ROUTE; + goto next; + } + + if (rte_pktmbuf_pkt_len(mbuf) > iface->mtu) { + m = rte_pktmbuf_mtod(mbuf, struct rte_mpls_hdr *); + depth = 0; + while (depth < GR_MPLS_MAX_STACK_DEPTH && !m->bs) { + m++; + depth++; + } + ver = ((const uint8_t *)(m + 1))[0] >> 4; + if (ver == 6) { + edge = TRANSIT_PKT_TOO_BIG; + } else if (ver == 4) { + ip4 = (const struct rte_ipv4_hdr *)(m + 1); + edge = (ip4->fragment_offset + & rte_cpu_to_be_16(RTE_IPV4_HDR_DF_FLAG)) ? + TRANSIT_FRAG_NEEDED : + MTU_EXCEEDED; + } else { + edge = MTU_EXCEEDED; + } + goto next; + } + + if (l3->state != GR_NH_S_REACHABLE) { + mpls_hold_mbuf_data(mbuf)->mpls_nh = nh; + l3_mbuf_data(mbuf)->nh = via_nh; + mbuf->packet_type |= RTE_PTYPE_TUNNEL_MPLS_IN_GRE; + edge = HOLD; + goto next; + } + + eth_data = eth_output_mbuf_data(mbuf); + eth_data->dst = l3->mac; + if (mbuf->packet_type & RTE_PTYPE_L3_IPV4) + eth_data->ether_type = RTE_BE16(RTE_ETHER_TYPE_IPV4); + else if (mbuf->packet_type & RTE_PTYPE_L3_IPV6) + eth_data->ether_type = RTE_BE16(RTE_ETHER_TYPE_IPV6); + else + eth_data->ether_type = RTE_BE16(RTE_ETHER_TYPE_MPLS); + mbuf_data(mbuf)->iface = iface; + edge = ETH_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static struct rte_node_register mpls_output_node_reg = { + .name = "mpls_output", + + .process = mpls_output_process, + + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ETH_OUTPUT] = "eth_output", + [HOLD] = "ip_hold", + [NO_ROUTE] = "mpls_output_no_route", + [TRANSIT_FRAG_NEEDED] = "mpls_output_frag_needed", + [TRANSIT_PKT_TOO_BIG] = "mpls_output_pkt_too_big", + [MTU_EXCEEDED] = "mpls_output_mtu_exceeded", + }, +}; + +static void mpls_output_register(void) { + mpls_output_ctrl = gr_control_input_register_handler("mpls_output"); +} + +static struct gr_node_info info = { + .node = &mpls_output_node_reg, + .type = GR_NODE_T_L3, + .register_callback = mpls_output_register, +}; + +GR_NODE_REGISTER(info); + +GR_DROP_REGISTER(mpls_output_no_route); +GR_DROP_REGISTER(mpls_output_mtu_exceeded); diff --git a/modules/mpls/datapath/mpls_push.c b/modules/mpls/datapath/mpls_push.c new file mode 100644 index 000000000..5749bb177 --- /dev/null +++ b/modules/mpls/datapath/mpls_push.c @@ -0,0 +1,137 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "graph.h" +#include "ip4_datapath.h" +#include "ip6_datapath.h" +#include "l3.h" +#include "mbuf.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include +#include + +enum { + MPLS_OUTPUT = 0, + NO_HEADROOM, + FRAG_NEEDED, + PKT_TOO_BIG, + MTU_EXCEEDED, + EDGE_COUNT, +}; + +static uint16_t +mpls_push_process(struct rte_graph *graph, struct rte_node *node, void **objs, uint16_t nb_objs) { + const struct nexthop_info_mpls *info; + struct rte_mpls_hdr *mpls; + const struct iface *iface; + struct rte_ipv6_hdr *ip6; + struct rte_ipv4_hdr *ip4; + const struct nexthop *nh; + struct rte_mbuf *mbuf; + uint16_t overhead; + rte_edge_t edge; + uint8_t ttl; + uint8_t tc; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + nh = l3_mbuf_data(mbuf)->nh; + info = nexthop_info_mpls(nh); + + if (info->n_labels == 0 || info->via_nh == NULL) { + edge = NO_HEADROOM; + goto next; + } + + iface = iface_from_id(info->via_nh->iface_id); + if (iface == NULL) { + edge = NO_HEADROOM; + goto next; + } + + overhead = info->n_labels * sizeof(struct rte_mpls_hdr); + if (rte_pktmbuf_pkt_len(mbuf) + overhead > iface->mtu) { + if (mbuf->packet_type & RTE_PTYPE_L3_IPV6) { + edge = PKT_TOO_BIG; + } else { + ip4 = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + edge = (ip4->fragment_offset + & rte_cpu_to_be_16(RTE_IPV4_HDR_DF_FLAG)) ? + FRAG_NEEDED : + MTU_EXCEEDED; + } + goto next; + } + + if (mbuf->packet_type & RTE_PTYPE_L3_IPV4) { + ip4 = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + ttl = ip4->time_to_live; + tc = ip4->type_of_service >> 5; + } else { + ip6 = rte_pktmbuf_mtod(mbuf, struct rte_ipv6_hdr *); + ttl = ip6->hop_limits; + tc = (rte_be_to_cpu_32(ip6->vtc_flow) >> 25) & 0x7; + } + + if (info->ttl) + ttl = info->ttl; + + mpls = gr_mbuf_prepend(mbuf, mpls, (info->n_labels - 1) * sizeof(*mpls)); + if (unlikely(mpls == NULL)) { + edge = NO_HEADROOM; + goto next; + } + + for (uint8_t j = 0; j < info->n_labels; j++) { + mpls_hdr_set_label(&mpls[j], info->labels[j]); + mpls[j].tc = tc; + mpls[j].bs = (j == info->n_labels - 1) ? 1 : 0; + mpls[j].ttl = ttl; + } + + l3_mbuf_data(mbuf)->nh = nh; + mbuf->packet_type = RTE_PTYPE_TUNNEL_MPLS_IN_GRE; + edge = MPLS_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static void mpls_push_register(void) { + ip_output_register_nexthop_type(GR_NH_T_MPLS, "mpls_push"); + ip6_output_register_nexthop_type(GR_NH_T_MPLS, "mpls_push"); +} + +static struct rte_node_register mpls_push_node = { + .name = "mpls_push", + + .process = mpls_push_process, + + .nb_edges = EDGE_COUNT, + .next_nodes = { + [MPLS_OUTPUT] = "mpls_output", + [NO_HEADROOM] = "error_no_headroom", + [FRAG_NEEDED] = "mpls_push_frag_needed", + [PKT_TOO_BIG] = "mpls_push_pkt_too_big", + [MTU_EXCEEDED] = "mpls_push_mtu_exceeded", + }, +}; + +static struct gr_node_info info = { + .node = &mpls_push_node, + .type = GR_NODE_T_L3, + .register_callback = mpls_push_register, +}; + +GR_NODE_REGISTER(info); + +GR_DROP_REGISTER(mpls_push_mtu_exceeded); diff --git a/modules/mpls/meson.build b/modules/mpls/meson.build index 7d9a19856..870666fd5 100644 --- a/modules/mpls/meson.build +++ b/modules/mpls/meson.build @@ -3,3 +3,4 @@ subdir('api') subdir('control') +subdir('datapath') From 52e1c2fc1e65a7e52b6416e72b129a72cf68f59e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Mon, 29 Jun 2026 14:13:16 +0200 Subject: [PATCH 06/11] mpls: add CLI commands MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Register grcli commands for MPLS nexthop creation and label route management. The nexthop command supports push, pop, optional TTL override, and explicit payload type for egress disposition. A formatter is registered so that nexthop show displays MPLS-specific fields. Label route commands use a separate CLI context since label routes are keyed by numeric labels and cannot reuse the existing IP route CLI dispatch. Event printers for label route add/del notifications are also registered. Signed-off-by: Matej Mužila --- modules/mpls/cli/label.c | 226 +++++++++++++++++++++++++++++++++++ modules/mpls/cli/meson.build | 8 ++ modules/mpls/cli/nexthop.c | 199 ++++++++++++++++++++++++++++++ modules/mpls/meson.build | 1 + 4 files changed, 434 insertions(+) create mode 100644 modules/mpls/cli/label.c create mode 100644 modules/mpls/cli/meson.build create mode 100644 modules/mpls/cli/nexthop.c diff --git a/modules/mpls/cli/label.c b/modules/mpls/cli/label.c new file mode 100644 index 000000000..a74b52d6b --- /dev/null +++ b/modules/mpls/cli/label.c @@ -0,0 +1,226 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "cli.h" +#include "cli_event.h" +#include "cli_iface.h" +#include "display.h" + +#include +#include +#include + +#include + +#include + +static cmd_status_t mpls_route_add(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_mpls_label_route_add_req req; + + req = (struct gr_mpls_label_route_add_req) { + .exist_ok = true, + .origin = GR_NH_ORIGIN_STATIC, + }; + + if (arg_u32(p, "LABEL", &req.in_label) < 0) + return CMD_ERROR; + if (arg_u32(p, "ID", &req.nh_id) < 0) + return CMD_ERROR; + if (arg_vrf(c, p, "VRF", &req.vrf_id) < 0) + return CMD_ERROR; + + if (gr_api_client_send_recv(c, GR_MPLS_LABEL_ROUTE_ADD, sizeof(req), &req, NULL) < 0) + return CMD_ERROR; + + return CMD_SUCCESS; +} + +static cmd_status_t mpls_route_del(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_mpls_label_route_del_req req; + + req = (struct gr_mpls_label_route_del_req) {.missing_ok = true}; + + if (arg_u32(p, "LABEL", &req.in_label) < 0) + return CMD_ERROR; + if (arg_vrf(c, p, "VRF", &req.vrf_id) < 0) + return CMD_ERROR; + + if (gr_api_client_send_recv(c, GR_MPLS_LABEL_ROUTE_DEL, sizeof(req), &req, NULL) < 0) + return CMD_ERROR; + + return CMD_SUCCESS; +} + +static cmd_status_t mpls_route_get(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_mpls_label_route_get_req req; + const struct gr_mpls_label_route *resp; + struct gr_object *o; + void *resp_ptr; + + req = (struct gr_mpls_label_route_get_req) {0}; + resp_ptr = NULL; + + if (arg_u32(p, "LABEL", &req.in_label) < 0) + return CMD_ERROR; + if (arg_vrf(c, p, "VRF", &req.vrf_id) < 0) + return CMD_ERROR; + + if (gr_api_client_send_recv(c, GR_MPLS_LABEL_ROUTE_GET, sizeof(req), &req, &resp_ptr) < 0) + return CMD_ERROR; + + resp = resp_ptr; + o = gr_object_new(NULL); + gr_object_field(o, "vrf", 0, "%s", iface_name_from_id(c, resp->vrf_id)); + gr_object_field(o, "label", 0, "%u", resp->in_label); + gr_object_field(o, "nexthop_id", 0, "%u", resp->nh_id); + gr_object_field(o, "origin", 0, "%s", gr_nh_origin_name(resp->origin)); + gr_object_free(o); + free(resp_ptr); + + return CMD_SUCCESS; +} + +static cmd_status_t mpls_route_list(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_mpls_label_route_list_req req; + const struct gr_mpls_label_route *route; + struct gr_table *table; + uint16_t max_routes; + uint16_t vrf_id; + int ret; + + vrf_id = GR_VRF_ID_UNDEF; + max_routes = 1000; + + if (arg_str(p, "VRF") != NULL && arg_vrf(c, p, "VRF", &vrf_id) < 0) + return CMD_ERROR; + if (arg_u16(p, "MAX", &max_routes) < 0 && errno != ENOENT) + return CMD_ERROR; + + req = (struct gr_mpls_label_route_list_req) { + .vrf_id = vrf_id, + .max_count = max_routes, + }; + + table = gr_table_new(); + gr_table_column(table, "VRF", GR_DISP_LEFT); + gr_table_column(table, "LABEL", GR_DISP_RIGHT); + gr_table_column(table, "NEXTHOP_ID", GR_DISP_RIGHT); + gr_table_column(table, "ORIGIN", GR_DISP_LEFT); + + gr_api_client_stream_foreach (route, ret, c, GR_MPLS_LABEL_ROUTE_LIST, sizeof(req), &req) { + gr_table_cell(table, 0, "%s", iface_name_from_id(c, route->vrf_id)); + gr_table_cell(table, 1, "%u", route->in_label); + gr_table_cell(table, 2, "%u", route->nh_id); + gr_table_cell(table, 3, "%s", gr_nh_origin_name(route->origin)); + + if (gr_table_print_row(table) < 0) + break; + } + + gr_table_free(table); + + if (ret < 0 && errno == EXFULL) { + warnf("more routes not displayed"); + ret = 0; + } + + return ret < 0 ? CMD_ERROR : CMD_SUCCESS; +} + +#define MPLS_CTX(root) CLI_CONTEXT(root, CTX_ARG("mpls", "MPLS label switching.")) +#define MPLS_ROUTE_CTX(root) CLI_CONTEXT(MPLS_CTX(root), CTX_ARG("route", "Label routes.")) + +static int ctx_init(struct ec_node *root) { + int ret; + + ret = CLI_COMMAND( + MPLS_ROUTE_CTX(root), + "add LABEL nexthop id ID [vrf VRF]", + mpls_route_add, + "Add a label route.", + with_help("MPLS label value (0-1048575).", ec_node_uint("LABEL", 0, 1048575, 10)), + with_help("Nexthop user ID.", ec_node_uint("ID", 1, UINT32_MAX - 1, 10)), + with_help("L3 routing domain name.", ec_node_dyn("VRF", complete_vrf_names, NULL)) + ); + if (ret < 0) + return ret; + ret = CLI_COMMAND( + MPLS_ROUTE_CTX(root), + "del LABEL [vrf VRF]", + mpls_route_del, + "Delete a label route.", + with_help("MPLS label value.", ec_node_uint("LABEL", 0, 1048575, 10)), + with_help("L3 routing domain name.", ec_node_dyn("VRF", complete_vrf_names, NULL)) + ); + if (ret < 0) + return ret; + ret = CLI_COMMAND( + MPLS_ROUTE_CTX(root), + "get LABEL [vrf VRF]", + mpls_route_get, + "Get a label route.", + with_help("MPLS label value.", ec_node_uint("LABEL", 0, 1048575, 10)), + with_help("L3 routing domain name.", ec_node_dyn("VRF", complete_vrf_names, NULL)) + ); + if (ret < 0) + return ret; + ret = CLI_COMMAND( + MPLS_ROUTE_CTX(root), + "[show] [(vrf VRF),(max MAX)]", + mpls_route_list, + "Show label routes.", + with_help( + "Max. number of routes to display (default 1000, use 0 for unlimited).", + ec_node_uint("MAX", 0, UINT16_MAX, 10) + ), + with_help("L3 routing domain name.", ec_node_dyn("VRF", complete_vrf_names, NULL)) + ); + if (ret < 0) + return ret; + + return 0; +} + +static void mpls_event_print(uint32_t event, const void *obj) { + const struct gr_mpls_label_route *r = obj; + const char *action; + + switch (event) { + case GR_EVENT_MPLS_ROUTE_ADD: + action = "add"; + break; + case GR_EVENT_MPLS_ROUTE_DEL: + action = "del"; + break; + default: + action = "?"; + break; + } + + printf("mpls route %s: vrf=%s label=%u nh_id=%u origin=%s\n", + action, + iface_name_from_id(NULL, r->vrf_id), + r->in_label, + r->nh_id, + gr_nh_origin_name(r->origin)); +} + +static struct cli_event_printer printer = { + .name = "mpls_route", + .print = mpls_event_print, + .ev_count = 2, + .ev_types = { + GR_EVENT_MPLS_ROUTE_ADD, + GR_EVENT_MPLS_ROUTE_DEL, + }, +}; + +static struct cli_context ctx = { + .name = "mpls", + .init = ctx_init, +}; + +static void __attribute__((constructor, used)) init(void) { + cli_context_register(&ctx); + cli_event_printer_register(&printer); +} diff --git a/modules/mpls/cli/meson.build b/modules/mpls/cli/meson.build new file mode 100644 index 000000000..d28bae09e --- /dev/null +++ b/modules/mpls/cli/meson.build @@ -0,0 +1,8 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +cli_src += files( + 'label.c', + 'nexthop.c', +) +cli_inc += include_directories('.') diff --git a/modules/mpls/cli/nexthop.c b/modules/mpls/cli/nexthop.c new file mode 100644 index 000000000..87fe1f267 --- /dev/null +++ b/modules/mpls/cli/nexthop.c @@ -0,0 +1,199 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "cli.h" +#include "cli_iface.h" +#include "cli_nexthop.h" +#include "display.h" + +#include +#include +#include + +#include + +#include + +static cmd_status_t nh_mpls_add(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_nexthop_info_mpls *info; + struct gr_nh_add_req *req; + const struct ec_pnode *n; + unsigned long val; + cmd_status_t ret; + const char *str; + size_t len; + + req = NULL; + ret = CMD_ERROR; + + len = sizeof(*req) + sizeof(*info); + req = calloc(1, len); + if (req == NULL) + goto out; + + req->exist_ok = true; + req->nh.type = GR_NH_T_MPLS; + req->nh.origin = GR_NH_ORIGIN_STATIC; + + if (arg_u32(p, "ID", &req->nh.nh_id) < 0 && errno != ENOENT) + goto out; + if (arg_iface(c, p, "IFACE", GR_IFACE_TYPE_UNDEF, &req->nh.iface_id) < 0) + goto out; + + info = (struct gr_nexthop_info_mpls *)req->nh.info; + + switch (arg_ip4(p, "VIA", &info->via.ipv4)) { + case 0: + info->via.af = GR_AF_IP4; + break; + case -EINVAL: + if (arg_ip6(p, "VIA", &info->via.ipv6) < 0) + goto out; + info->via.af = GR_AF_IP6; + break; + default: + errno = EINVAL; + goto out; + } + + if (arg_u8(p, "TTL", &info->ttl) < 0 && errno != ENOENT) + goto out; + + if (arg_str(p, "ipv4") != NULL) + info->payload_af = GR_AF_IP4; + else if (arg_str(p, "ipv6") != NULL) + info->payload_af = GR_AF_IP6; + + n = ec_pnode_find(p, "LABEL"); + if (n != NULL) { + n = ec_pnode_get_parent(n); + if (n == NULL || ec_pnode_len(n) < 1) { + errno = EINVAL; + goto out; + } + if (ec_pnode_len(n) > GR_MPLS_MAX_LABELS) { + errno = E2BIG; + goto out; + } + for (n = ec_pnode_get_first_child(n); n != NULL; n = ec_pnode_next(n)) { + str = ec_strvec_val(ec_pnode_get_strvec(n), 0); + val = strtoul(str, NULL, 10); + if (val > GR_MPLS_LABEL_MAX) { + errno = EINVAL; + goto out; + } + info->labels[info->n_labels++] = (uint32_t)val; + } + } + + if (gr_api_client_send_recv(c, GR_NH_ADD, len, req, NULL) < 0) + goto out; + + ret = CMD_SUCCESS; +out: + free(req); + return ret; +} + +static void add_columns_mpls(struct gr_table *table) { + gr_table_column(table, "VIA", GR_DISP_LEFT); + gr_table_column(table, "LABELS", GR_DISP_STR_ARRAY); + gr_table_column(table, "TTL", GR_DISP_RIGHT); +} + +static void fill_table_mpls(struct gr_table *table, unsigned start_col, const void *nexthop_info) { + const struct gr_nexthop_info_mpls *info = nexthop_info; + char buf[256]; + ssize_t n; + + buf[0] = '\0'; + n = 0; + + if (info->via.af == GR_AF_IP4) + gr_table_cell(table, start_col, IP4_F, &info->via.ipv4); + else if (info->via.af == GR_AF_IP6) + gr_table_cell(table, start_col, IP6_F, &info->via.ipv6); + else + gr_table_cell(table, start_col, "-"); + + for (uint8_t i = 0; i < info->n_labels; i++) { + SAFE_BUF(snprintf, sizeof(buf), "%s%u", i > 0 ? " " : "", info->labels[i]); + if (sizeof(buf) - n < 20) { + SAFE_BUF(snprintf, sizeof(buf), " ... (%u more)", info->n_labels - i - 1); + break; + } + } +err: + if (info->n_labels > 0 && n > 0) + gr_table_cell(table, start_col + 1, "%s", buf); + else + gr_table_cell(table, start_col + 1, "(pop)"); + + if (info->ttl > 0) + gr_table_cell(table, start_col + 2, "%u", info->ttl); + else + gr_table_cell(table, start_col + 2, "-"); +} + +static void fill_object_mpls(struct gr_object *o, const void *nexthop_info) { + const struct gr_nexthop_info_mpls *info = nexthop_info; + + if (info->via.af == GR_AF_IP4) + gr_object_field(o, "via", 0, IP4_F, &info->via.ipv4); + else if (info->via.af == GR_AF_IP6) + gr_object_field(o, "via", 0, IP6_F, &info->via.ipv6); + + if (info->n_labels > 0) { + gr_object_array_open(o, "labels"); + for (uint8_t i = 0; i < info->n_labels; i++) + gr_object_array_item(o, GR_DISP_INT, "%u", info->labels[i]); + gr_object_array_close(o); + } else { + gr_object_field(o, "action", 0, "pop"); + } + + if (info->ttl > 0) + gr_object_field(o, "ttl", GR_DISP_INT, "%u", info->ttl); + if (info->payload_af != GR_AF_UNSPEC) + gr_object_field(o, "payload", 0, "%s", gr_af_name(info->payload_af)); +} + +static struct cli_nexthop_formatter mpls_formatter = { + .name = "mpls", + .type = GR_NH_T_MPLS, + .add_columns = add_columns_mpls, + .fill_table = fill_table_mpls, + .fill_object = fill_object_mpls, +}; + +static int ctx_init(struct ec_node *root) { + int ret; + + ret = CLI_COMMAND( + NEXTHOP_ADD_CTX(root), + "mpls iface IFACE via VIA [labels LABEL+] [(id ID),(ttl TTL),(payload ipv4|ipv6)]", + nh_mpls_add, + "Add an MPLS nexthop.", + with_help("Output interface.", ec_node_dyn("IFACE", complete_iface_names, NULL)), + with_help("Gateway IPv4/6 address.", ec_node_re("VIA", IP_ANY_RE)), + with_help("Output MPLS label (0-1048575).", ec_node_uint("LABEL", 0, 1048575, 10)), + with_help("Nexthop ID.", ec_node_uint("ID", 1, UINT32_MAX - 1, 10)), + with_help("Initial TTL (1-255, 0=copy).", ec_node_uint("TTL", 1, 255, 10)), + with_help("IPv4 payload.", ec_node_str("ipv4", "ipv4")), + with_help("IPv6 payload.", ec_node_str("ipv6", "ipv6")) + ); + if (ret < 0) + return ret; + + return 0; +} + +static struct cli_context ctx = { + .name = "mpls_nexthop", + .init = ctx_init, +}; + +static void __attribute__((constructor, used)) init(void) { + cli_context_register(&ctx); + cli_nexthop_formatter_register(&mpls_formatter); +} diff --git a/modules/mpls/meson.build b/modules/mpls/meson.build index 870666fd5..8bc77a9c3 100644 --- a/modules/mpls/meson.build +++ b/modules/mpls/meson.build @@ -2,5 +2,6 @@ # Copyright (c) 2025 Matej Muzila subdir('api') +subdir('cli') subdir('control') subdir('datapath') From b53c789e0028abb96f0451e356e4ae1990f59d98 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Thu, 9 Jul 2026 10:25:28 +0200 Subject: [PATCH 07/11] mpls: add smoke tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cover the main MPLS datapath operations for both IPv4 and IPv6: label push and pop, label swap, explicit null (labels 0 and 2), and multi-label push. TTL propagation across push, swap, and pop is tested separately. PMTU handling is verified with pkt_too_big (IPv6) and frag_needed (IPv4 DF) tests. A hold test verifies that packets queued during ARP resolution are correctly forwarded once the neighbor is resolved. A separate CLI test exercises nexthop and label route CRUD through grcli. Tests that send MPLS-encapsulated traffic from Linux namespaces pre-resolve ARP with a plain IP ping before the MPLS test to avoid a race between the iptunnel encap path and neighbor resolution. Signed-off-by: Matej Mužila --- smoke/mpls_cli_test.sh | 56 +++++++++++++++++++++++++++++ smoke/mpls_explicit_null6_test.sh | 38 ++++++++++++++++++++ smoke/mpls_explicit_null_test.sh | 38 ++++++++++++++++++++ smoke/mpls_frag_needed_test.sh | 53 +++++++++++++++++++++++++++ smoke/mpls_hold_test.sh | 42 ++++++++++++++++++++++ smoke/mpls_multi_label_push_test.sh | 46 ++++++++++++++++++++++++ smoke/mpls_pkt_too_big_test.sh | 53 +++++++++++++++++++++++++++ smoke/mpls_pop6_test.sh | 42 ++++++++++++++++++++++ smoke/mpls_pop_payload_test.sh | 42 ++++++++++++++++++++++ smoke/mpls_pop_test.sh | 42 ++++++++++++++++++++++ smoke/mpls_push6_test.sh | 43 ++++++++++++++++++++++ smoke/mpls_push_test.sh | 43 ++++++++++++++++++++++ smoke/mpls_swap_test.sh | 47 ++++++++++++++++++++++++ smoke/mpls_ttl_test.sh | 47 ++++++++++++++++++++++++ 14 files changed, 632 insertions(+) create mode 100755 smoke/mpls_cli_test.sh create mode 100755 smoke/mpls_explicit_null6_test.sh create mode 100755 smoke/mpls_explicit_null_test.sh create mode 100755 smoke/mpls_frag_needed_test.sh create mode 100755 smoke/mpls_hold_test.sh create mode 100755 smoke/mpls_multi_label_push_test.sh create mode 100755 smoke/mpls_pkt_too_big_test.sh create mode 100755 smoke/mpls_pop6_test.sh create mode 100755 smoke/mpls_pop_payload_test.sh create mode 100755 smoke/mpls_pop_test.sh create mode 100755 smoke/mpls_push6_test.sh create mode 100755 smoke/mpls_push_test.sh create mode 100755 smoke/mpls_swap_test.sh create mode 100755 smoke/mpls_ttl_test.sh diff --git a/smoke/mpls_cli_test.sh b/smoke/mpls_cli_test.sh new file mode 100755 index 000000000..8b9173ab7 --- /dev/null +++ b/smoke/mpls_cli_test.sh @@ -0,0 +1,56 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +grcli address add 172.16.0.1/24 iface p0 + +netns_add n0 +move_to_netns x-p0 n0 +ip -n n0 addr add 172.16.0.2/24 dev x-p0 + +# create MPLS nexthop with single label +grcli nexthop add mpls iface p0 via 172.16.0.2 labels 100 id 10 + +# verify nexthop is visible +grcli nexthop show id 10 + +# create MPLS nexthop with multiple labels +grcli nexthop add mpls iface p0 via 172.16.0.2 labels 100 200 300 id 20 + +# verify multi-label nexthop +grcli nexthop show id 20 + +# create pop nexthop (no labels) +grcli nexthop add mpls iface p0 via 172.16.0.2 id 30 + +# verify pop nexthop +grcli nexthop show id 30 + +# list MPLS nexthops +grcli nexthop show type mpls + +# add label route +grcli mpls route add 500 nexthop id 10 + +# get label route +grcli mpls route get 500 + +# list label routes +grcli mpls route show + +# idempotent add (exist_ok) +grcli mpls route add 500 nexthop id 10 + +# delete label route +grcli mpls route del 500 + +# idempotent delete (missing_ok) +grcli mpls route del 500 + +# cleanup nexthops +grcli nexthop del 10 +grcli nexthop del 20 +grcli nexthop del 30 diff --git a/smoke/mpls_explicit_null6_test.sh b/smoke/mpls_explicit_null6_test.sh new file mode 100755 index 000000000..2b83ab3f4 --- /dev/null +++ b/smoke/mpls_explicit_null6_test.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 2001:db8:0::1/64 iface p0 +grcli address add 2001:db8:1::1/64 iface p1 + +# IPv6 route for the decapsulated packet (no MPLS nexthop needed) +grcli route add fd00::/64 via 2001:db8:1::2 + +# return path +grcli route add fd00:0:0:1::/64 via 2001:db8:0::2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 2001:db8:$n::2/64 dev $p +done + +# n0: push IPv6 Explicit NULL (label 2) +ip -n n0 addr add fd00:0:0:1::1/128 dev lo +ip -n n0 route add fd00::/64 encap mpls 2 via 2001:db8:0::1 dev x-p0 + +# n1: plain IPv6 receiver +ip -n n1 addr add fd00::1/128 dev lo +ip -n n1 route add default via 2001:db8:1::1 + +# resolve NDP before MPLS test +ip netns exec n0 ping6 -c1 -n 2001:db8:0::1 + +# n0 pushes MPLS(label=2) -> grout strips and routes via ip6_input -> n1 receives plain IPv6 +ip netns exec n0 ping6 -i0.01 -c3 -n fd00::1 diff --git a/smoke/mpls_explicit_null_test.sh b/smoke/mpls_explicit_null_test.sh new file mode 100755 index 000000000..c7ae24825 --- /dev/null +++ b/smoke/mpls_explicit_null_test.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# IP route for the decapsulated packet (no MPLS nexthop needed) +grcli route add 10.0.1.0/24 via 172.16.1.2 + +# return path +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push explicit null (label 0) +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 0 via 172.16.0.1 dev x-p0 + +# n1: plain IP receiver +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# n0 pushes MPLS(label=0) -> grout strips and routes via ip_input -> n1 receives plain IP +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_frag_needed_test.sh b/smoke/mpls_frag_needed_test.sh new file mode 100755 index 000000000..7e910cbef --- /dev/null +++ b/smoke/mpls_frag_needed_test.sh @@ -0,0 +1,53 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: push label 100 (4 bytes overhead), forward via n1 +# Effective MTU = iface_mtu(1500) - 1*4 = 1496 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 100 id 42 + +# IP route pointing to MPLS nexthop +grcli route add 10.0.1.0/24 via id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: plain IP sender +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add default via 172.16.0.1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 100 dev lo +ip -n n1 route add default via 172.16.1.1 + +# verify basic connectivity first +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 + +# send oversized IPv4 packet with DF set: payload=1477 -> IP=1497 -> with label=1501 > 1500 +# ping -s sets payload size; total IP = payload + 28 (ICMP+IP headers) +# we need IP size > 1496 (effective MTU), so payload > 1468 +# use payload=1470 -> IP=1498 > 1496 +output=$(ip netns exec n0 ping -c1 -W2 -s1470 -M do -n 10.0.1.1 2>&1 || true) +echo "$output" +echo "$output" | grep -qi "frag needed\|message too big\|mtu\|unreachable" \ + || fail "expected ICMP Frag Needed for oversized DF packet, got: $output" + +echo "ICMP Frag Needed correctly received for oversized DF packet" diff --git a/smoke/mpls_hold_test.sh b/smoke/mpls_hold_test.sh new file mode 100755 index 000000000..bc3275142 --- /dev/null +++ b/smoke/mpls_hold_test.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: push label 100, forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 100 id 42 + +# IP route pointing to MPLS nexthop +grcli route add 10.0.1.0/24 via id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: plain IP sender +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add default via 172.16.0.1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 100 dev lo +ip -n n1 route add default via 172.16.1.1 + +# no ARP pre-warm: first packets will hit the hold path while ARP resolves +# use longer timeout (-W5) and several packets so ARP resolves during the burst +ip netns exec n0 ping -i0.1 -c10 -W5 -n 10.0.1.1 diff --git a/smoke/mpls_multi_label_push_test.sh b/smoke/mpls_multi_label_push_test.sh new file mode 100755 index 000000000..c612fd67b --- /dev/null +++ b/smoke/mpls_multi_label_push_test.sh @@ -0,0 +1,46 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: push two labels (outer=200 inner=100), forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 200 100 id 42 + +# IP route pointing to multi-label MPLS nexthop +grcli route add 10.0.1.0/24 via id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: plain IP sender +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add default via 172.16.0.1 + +# n1: MPLS receiver, pop outer label 200 then inner label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip netns exec n1 sysctl -wq net.mpls.conf.lo.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +# pop outer 200 via lo (recirculates as [100,IP] to lo), then pop inner 100 +ip -n n1 -f mpls route add 200 dev lo +ip -n n1 -f mpls route add 100 dev lo + +# return path from n1 +ip -n n1 route add default via 172.16.1.1 + +# plain IP -> grout pushes MPLS(200,100) -> n1 pops both labels +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_pkt_too_big_test.sh b/smoke/mpls_pkt_too_big_test.sh new file mode 100755 index 000000000..c63be29e8 --- /dev/null +++ b/smoke/mpls_pkt_too_big_test.sh @@ -0,0 +1,53 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 2001:db8:0::1/64 iface p0 +grcli address add 2001:db8:1::1/64 iface p1 + +# MPLS nexthop: push label 100 (4 bytes overhead), forward via n1 +# Effective MTU = iface_mtu(1500) - 1*4 = 1496 +grcli nexthop add mpls iface p1 via 2001:db8:1::2 labels 100 id 42 + +# IPv6 route pointing to MPLS nexthop +grcli route add fd00::/64 via id 42 + +# return path (plain IPv6) +grcli route add fd00:0:0:1::/64 via 2001:db8:0::2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 2001:db8:$n::2/64 dev $p +done + +# n0: plain IPv6 sender +ip -n n0 addr add fd00:0:0:1::1/128 dev lo +ip -n n0 route add default via 2001:db8:0::1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add fd00::1/128 dev lo +ip -n n1 -f mpls route add 100 dev lo +ip -n n1 route add default via 2001:db8:1::1 + +# verify basic connectivity first +ip netns exec n0 ping6 -i0.01 -c3 -n fd00::1 + +# send oversized IPv6 packet: payload=1450 -> IP=1490+40=1530 > 1496 effective MTU +# IPv6 always has DF semantics; ICMPv6 Packet Too Big must be returned +# ping6 -s sets data size; total IPv6 = data + 8 (ICMPv6) + 40 (IPv6 header) +# need IPv6 size > 1496, so data > 1448; use data=1450 -> IPv6=1498 > 1496 +output=$(ip netns exec n0 ping6 -c1 -W2 -s1450 -n fd00::1 2>&1 || true) +echo "$output" +echo "$output" | grep -qi "too big\|packet too big\|mtu\|unreachable" \ + || fail "expected ICMPv6 Packet Too Big for oversized IPv6 packet, got: $output" + +echo "ICMPv6 Packet Too Big correctly received for oversized IPv6 packet" diff --git a/smoke/mpls_pop6_test.sh b/smoke/mpls_pop6_test.sh new file mode 100755 index 000000000..6525ac03f --- /dev/null +++ b/smoke/mpls_pop6_test.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 2001:db8:0::1/64 iface p0 +grcli address add 2001:db8:1::1/64 iface p1 + +# MPLS nexthop: pop (no labels = PHP/pop), forward via n1 +grcli nexthop add mpls iface p1 via 2001:db8:1::2 id 42 + +# label route: incoming label 100 -> pop +grcli mpls route add 100 nexthop id 42 + +# return path (plain IPv6) +grcli route add fd00:0:0:1::/64 via 2001:db8:0::2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 2001:db8:$n::2/64 dev $p +done + +# n0: push label 100 for IPv6 traffic +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add fd00:0:0:1::1/128 dev lo +ip -n n0 route add fd00::/64 encap mpls 100 via 2001:db8:0::1 dev x-p0 + +# n1: plain IPv6 receiver +ip -n n1 addr add fd00::1/128 dev lo +ip -n n1 route add default via 2001:db8:1::1 + +# resolve NDP before MPLS test +ip netns exec n0 ping6 -c1 -n 2001:db8:0::1 + +# n0 pushes MPLS(100) -> grout pops -> n1 receives plain IPv6 +ip netns exec n0 ping6 -i0.01 -c3 -n fd00::1 diff --git a/smoke/mpls_pop_payload_test.sh b/smoke/mpls_pop_payload_test.sh new file mode 100755 index 000000000..64c892c88 --- /dev/null +++ b/smoke/mpls_pop_payload_test.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: pop with explicit payload type +grcli nexthop add mpls iface p1 via 172.16.1.2 id 42 payload ipv4 + +# label route: incoming label 100 -> pop +grcli mpls route add 100 nexthop id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push label 100 when sending to 10.0.1.0/24 +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 dev x-p0 + +# n1: plain IP receiver +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# n0 pushes MPLS(100) -> grout pops using explicit payload type -> n1 receives plain IP +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_pop_test.sh b/smoke/mpls_pop_test.sh new file mode 100755 index 000000000..89b8df967 --- /dev/null +++ b/smoke/mpls_pop_test.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: pop (no labels), forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 id 42 + +# label route: incoming label 100 -> pop +grcli mpls route add 100 nexthop id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push label 100 when sending to 10.0.1.0/24 +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 dev x-p0 + +# n1: plain IP receiver +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# n0 pushes MPLS(100) -> grout pops label, propagates TTL -> n1 receives plain IP +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_push6_test.sh b/smoke/mpls_push6_test.sh new file mode 100755 index 000000000..82699868a --- /dev/null +++ b/smoke/mpls_push6_test.sh @@ -0,0 +1,43 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 2001:db8:0::1/64 iface p0 +grcli address add 2001:db8:1::1/64 iface p1 + +# MPLS nexthop: push label 100 for IPv6 traffic, forward via n1 +grcli nexthop add mpls iface p1 via 2001:db8:1::2 labels 100 id 42 + +# IPv6 route pointing to MPLS nexthop +grcli route add fd00::/64 via id 42 + +# return path (plain IPv6) +grcli route add fd00:0:0:1::/64 via 2001:db8:0::2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 2001:db8:$n::2/64 dev $p +done + +# n0: plain IPv6 sender +ip -n n0 addr add fd00:0:0:1::1/128 dev lo +ip -n n0 route add default via 2001:db8:0::1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add fd00::1/128 dev lo +ip -n n1 -f mpls route add 100 dev lo + +# return path from n1 +ip -n n1 route add default via 2001:db8:1::1 + +# plain IPv6 -> grout pushes MPLS label -> n1 pops label +ip netns exec n0 ping6 -i0.01 -c3 -n fd00::1 diff --git a/smoke/mpls_push_test.sh b/smoke/mpls_push_test.sh new file mode 100755 index 000000000..69e37bf9f --- /dev/null +++ b/smoke/mpls_push_test.sh @@ -0,0 +1,43 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: push label 100, forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 100 id 42 + +# IP route pointing to MPLS nexthop +grcli route add 10.0.1.0/24 via id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: plain IP sender +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add default via 172.16.0.1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 100 dev lo + +# return path from n1 +ip -n n1 route add default via 172.16.1.1 + +# plain IP -> grout pushes MPLS label -> n1 pops label +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_swap_test.sh b/smoke/mpls_swap_test.sh new file mode 100755 index 000000000..9f8df4a25 --- /dev/null +++ b/smoke/mpls_swap_test.sh @@ -0,0 +1,47 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: swap to label 200, forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 200 id 42 + +# label route: incoming label 100 -> swap to 200 +grcli mpls route add 100 nexthop id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push label 100 when sending to 10.0.1.0/24 +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 dev x-p0 + +# n1: pop label 200 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 200 dev lo + +# return path from n1 +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# n0 pushes MPLS(100) -> grout swaps to MPLS(200) -> n1 pops +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_ttl_test.sh b/smoke/mpls_ttl_test.sh new file mode 100755 index 000000000..9653e7902 --- /dev/null +++ b/smoke/mpls_ttl_test.sh @@ -0,0 +1,47 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: swap to label 200, forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 200 id 42 + +# label route: incoming label 100 -> swap to 200 +grcli mpls route add 100 nexthop id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push label 100 +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 dev x-p0 + +# n1: pop label 200 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 200 dev lo +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# TTL=1: grout must drop the MPLS-encapped packet and not forward it +ip netns exec n0 ping -c1 -W1 -t1 -n 10.0.1.1 && fail "packet with TTL=1 should not reach destination" + +echo "TTL=1 packet correctly dropped" From c1639026773758a8acf5f597f42d1a6dc1b13812 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Fri, 18 Sep 2026 15:27:16 +0200 Subject: [PATCH 08/11] frr: add MPLS LSP and labeled nexthop support MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Handle DPLANE_OP_LSP_* by building a GR_NH_T_MPLS nexthop from the best NHLFE and installing a label route pointing at it. The nexthop id is derived deterministically from the incoming label, offset clear of both grout's allocation pool and FRR's nexthop group range to avoid collisions. Implicit-null maps to a nexthop with no imposed labels (PHP); explicit null (0 and 2) is kept on the wire. On delete, the label route is removed before the nexthop to avoid dangling references. Labeled nexthops from DPLANE_OP_NH_INSTALL (BGP-LU, L3VPN, SR-MPLS) are also turned into GR_NH_T_MPLS nexthops. Detection runs after the SRv6 checks to preserve existing SR behaviour. The grout-frr(7) man page is updated to document MPLS support. Signed-off-by: Matej Mužila --- docs/grout-frr.7.scdoc | 4 +- frr/rt_grout.c | 201 +++++++++++++++++++++++++++++++++++++++ frr/rt_grout.h | 3 + frr/zebra_dplane_grout.c | 74 ++++++++++++-- 4 files changed, 274 insertions(+), 8 deletions(-) diff --git a/docs/grout-frr.7.scdoc b/docs/grout-frr.7.scdoc index 1386a7025..bdcab0186 100644 --- a/docs/grout-frr.7.scdoc +++ b/docs/grout-frr.7.scdoc @@ -27,7 +27,9 @@ routes to grout, and grout notifies FRR of link and address changes. The following zebra dataplane operations are handled: - IPv4/IPv6 route install, update and delete -- Nexthop install, update and delete +- Nexthop install, update and delete (including labeled nexthops for + MPLS imposition, e.g. BGP-LU, L3VPN and SR-MPLS) +- MPLS LSP install, update and delete (e.g. from LDP) - IP address install and uninstall - MAC/FDB install and delete - VXLAN flood VTEP add and delete diff --git a/frr/rt_grout.c b/frr/rt_grout.c index 8c5bdb32b..6be466941 100644 --- a/frr/rt_grout.c +++ b/frr/rt_grout.c @@ -7,9 +7,11 @@ #include "rt_grout.h" #include +#include #include #include +#include #include #include #include @@ -17,8 +19,15 @@ #include #include #include +#include #include +// Synthesize a deterministic grout nexthop ID for an MPLS label route. Label +// routes are 1:1 with their incoming label, so the label doubles as the key. +// The 0x80000000 offset keeps these IDs clear of both grout's allocation pool +// (<= 1 << 17) and FRR's proto-NHG range (< ZEBRA_NHG_PROTO_UPPER, ~250M). +#define GROUT_MPLS_NH_ID(label) (0x80000000u | (uint32_t)(label)) + static inline bool is_selfroute(gr_nh_origin_t origin) { switch (origin) { case GR_NH_ORIGIN_ZEBRA: @@ -507,6 +516,20 @@ void grout_route6_change(bool new, struct gr_ip6_route *gr_r6, bool startup) { ); } +void grout_mpls_route_change(bool new, const struct gr_mpls_label_route *route, bool /*startup*/) { + gr_log_debug( + "%s in_label %u vrf %u nh_id %u", + new ? "add" : "del", + route->in_label, + route->vrf_id, + route->nh_id + ); + // MPLS route events from grout (e.g. statically added via grcli) are + // informational only — FRR is the label distribution protocol and owns + // the LFIB. A full ZAPI push would require lsp_add/del_nhlfe which is + // internal to zebra_mpls. +} + enum zebra_dplane_result grout_add_del_route(struct zebra_dplane_ctx *ctx) { union { struct gr_ip4_route_add_req r4_add; @@ -627,6 +650,177 @@ enum zebra_dplane_result grout_add_del_route(struct zebra_dplane_ctx *ctx) { return ZEBRA_DPLANE_REQUEST_SUCCESS; } +static inline gr_nh_origin_t lsptype2origin(enum lsp_types_t type) { + switch (type) { + case ZEBRA_LSP_STATIC: + return GR_NH_ORIGIN_ZSTATIC; + case ZEBRA_LSP_LDP: + return GR_NH_ORIGIN_LDP; + case ZEBRA_LSP_BGP: + return GR_NH_ORIGIN_BGP; + case ZEBRA_LSP_OSPF_SR: + return GR_NH_ORIGIN_OSPF; + case ZEBRA_LSP_ISIS_SR: + return GR_NH_ORIGIN_ISIS; + case ZEBRA_LSP_SHARP: + return GR_NH_ORIGIN_SHARP; + case ZEBRA_LSP_SRTE: + return GR_NH_ORIGIN_SRTE; + case ZEBRA_LSP_NONE: + case ZEBRA_LSP_EVPN: + default: + return GR_NH_ORIGIN_ZEBRA; + } +} + +// A lone implicit-null is penultimate hop popping, which grout +// represents as an MPLS nexthop with zero output labels, so it does not count +// as "having labels" here. +static bool nh_has_mpls_labels(const struct nexthop *nh) { + const struct mpls_label_stack *nhl = nh->nh_label; + + return nhl != NULL && nhl->num_labels > 0 + && !(nhl->num_labels == 1 && nhl->label[0] == MPLS_LABEL_IMPLICIT_NULL); +} + +// Populate an MPLS nexthop add request from a FRR nexthop carrying output +// labels. Returns 0 on success, -1 if the nexthop cannot be represented in +// grout. The caller must allocate req with room for a gr_nexthop_info_mpls. +static int grout_fill_mpls_nh( + struct gr_nh_add_req *req, + uint32_t nh_id, + gr_nh_origin_t origin, + const struct nexthop *nh +) { + struct gr_nexthop_info_mpls *mpls = (struct gr_nexthop_info_mpls *)req->nh.info; + + req->exist_ok = true; + req->nh.nh_id = nh_id; + req->nh.origin = origin; + req->nh.type = GR_NH_T_MPLS; + req->nh.vrf_id = vrf_frr_to_grout(nh->vrf_id); + req->nh.iface_id = ifindex_frr_to_grout(nh->ifindex); + + // grout resolves the outgoing L2 header through an L3 gateway, so an + // MPLS nexthop must carry one. Extract it from the FRR nexthop. + switch (nh->type) { + case NEXTHOP_TYPE_IPV4: + case NEXTHOP_TYPE_IPV4_IFINDEX: + mpls->via.af = GR_AF_IP4; + memcpy(&mpls->via.ipv4, &nh->gate.ipv4, sizeof(mpls->via.ipv4)); + break; + case NEXTHOP_TYPE_IPV6: + case NEXTHOP_TYPE_IPV6_IFINDEX: + mpls->via.af = GR_AF_IP6; + memcpy(&mpls->via.ipv6, &nh->gate.ipv6, sizeof(mpls->via.ipv6)); + break; + default: + gr_log_err("MPLS nexthop requires an IP gateway, type %u unsupported", nh->type); + return -1; + } + + // Copy the output label stack. A lone implicit-null (penultimate hop + // popping) leaves n_labels at 0, i.e. pop with no imposition. Explicit + // null (0/2) is a real label and stays on the wire. + if (nh_has_mpls_labels(nh)) { + const struct mpls_label_stack *nhl = nh->nh_label; + + if (nhl->num_labels > GR_MPLS_MAX_LABELS) { + gr_log_err("too many output labels: %u", nhl->num_labels); + return -1; + } + mpls->n_labels = nhl->num_labels; + for (uint8_t i = 0; i < nhl->num_labels; i++) + mpls->labels[i] = nhl->label[i]; + } + + mpls->ttl = 0; // copy TTL from the incoming label / payload + mpls->payload_af = GR_AF_UNSPEC; // auto-detect payload after pop + + return 0; +} + +enum zebra_dplane_result grout_add_del_lsp(struct zebra_dplane_ctx *ctx) { + struct gr_mpls_label_route_del_req del; + struct gr_mpls_label_route_add_req add; + const struct zebra_nhlfe *best; + struct gr_nh_del_req nh_del; + struct gr_nh_add_req *req; + mpls_label_t in_label; + gr_nh_origin_t origin; + uint32_t vrf_id; + uint32_t nh_id; + size_t len; + bool new; + + vrf_id = vrf_frr_to_grout(dplane_ctx_get_vrf(ctx)); + new = dplane_ctx_get_op(ctx) != DPLANE_OP_LSP_DELETE; + in_label = dplane_ctx_get_in_label(ctx); + nh_id = GROUT_MPLS_NH_ID(in_label); + + gr_log_debug("%s in_label %u vrf %u", new ? "add" : "del", in_label, vrf_id); + + if (in_label == MPLS_INVALID_LABEL || in_label > GR_MPLS_LABEL_MAX) { + gr_log_err("invalid in_label %u, skip", in_label); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + + if (!new) { + del = (struct gr_mpls_label_route_del_req) { + .vrf_id = vrf_id, + .in_label = in_label, + .missing_ok = true, + }; + nh_del = (struct gr_nh_del_req) {.missing_ok = true, .nh = {.nh_id = nh_id}}; + + // Drop the label route first so nothing references the nexthop, + // then remove the synthesized nexthop. + if (grout_client_send_recv(GR_MPLS_LABEL_ROUTE_DEL, sizeof(del), &del, NULL) < 0) + return ZEBRA_DPLANE_REQUEST_FAILURE; + if (grout_client_send_recv(GR_NH_DEL, sizeof(nh_del), &nh_del, NULL) < 0) + return ZEBRA_DPLANE_REQUEST_FAILURE; + return ZEBRA_DPLANE_REQUEST_SUCCESS; + } + + best = dplane_ctx_get_best_nhlfe(ctx); + if (best == NULL || best->nexthop == NULL) { + gr_log_err("LSP %u has no best nexthop, skip", in_label); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + + origin = lsptype2origin(best->type); + + len = sizeof(*req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + if (req == NULL) { + gr_log_err("calloc: %s", strerror(errno)); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + + if (grout_fill_mpls_nh(req, nh_id, origin, best->nexthop) < 0) { + free(req); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + + if (grout_client_send_recv(GR_NH_ADD, len, req, NULL) < 0) { + free(req); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + free(req); + + add = (struct gr_mpls_label_route_add_req) { + .vrf_id = vrf_id, + .in_label = in_label, + .nh_id = nh_id, + .origin = origin, + .exist_ok = true, + }; + if (grout_client_send_recv(GR_MPLS_LABEL_ROUTE_ADD, sizeof(add), &add, NULL) < 0) + return ZEBRA_DPLANE_REQUEST_FAILURE; + + return ZEBRA_DPLANE_REQUEST_SUCCESS; +} + static enum zebra_dplane_result grout_add_nexthop_group(struct zebra_dplane_ctx *ctx) { enum zebra_dplane_result ret = ZEBRA_DPLANE_REQUEST_SUCCESS; uint32_t nh_id = dplane_ctx_get_nhe_id(ctx); @@ -702,6 +896,9 @@ grout_add_nexthop(uint32_t nh_id, gr_nh_origin_t origin, const struct nexthop *n len += sizeof(*sr6) + nh->nh_srv6->seg6_segs->num_segs * sizeof(sr6->seglist[0]); type = GR_NH_T_SR6_OUTPUT; + } else if (nh_has_mpls_labels(nh)) { + len += sizeof(struct gr_nexthop_info_mpls); + type = GR_NH_T_MPLS; } else { len += sizeof(*l3); type = GR_NH_T_L3; @@ -731,6 +928,10 @@ grout_add_nexthop(uint32_t nh_id, gr_nh_origin_t origin, const struct nexthop *n req->nh.iface_id = ifindex_frr_to_grout(nh->ifindex); switch (type) { + case GR_NH_T_MPLS: + if (grout_fill_mpls_nh(req, nh_id, origin, nh) < 0) + goto out; + break; case GR_NH_T_L3: // For L3 nexthops in VRFs with an L3VNI, redirect the iface from // the VRF (SVI in FRR's model) to the VXLAN interface. Grout diff --git a/frr/rt_grout.h b/frr/rt_grout.h index a6f08bfdd..aafdde20c 100644 --- a/frr/rt_grout.h +++ b/frr/rt_grout.h @@ -6,12 +6,15 @@ #include #include #include +#include #include void grout_route4_change(bool new, struct gr_ip4_route *gr_r4, bool startup); void grout_route6_change(bool new, struct gr_ip6_route *gr_r6, bool startup); +void grout_mpls_route_change(bool new, const struct gr_mpls_label_route *route, bool /*startup*/); enum zebra_dplane_result grout_add_del_route(struct zebra_dplane_ctx *ctx); +enum zebra_dplane_result grout_add_del_lsp(struct zebra_dplane_ctx *ctx); enum zebra_dplane_result grout_add_del_nexthop(struct zebra_dplane_ctx *ctx); void grout_nexthop_change(bool new, struct gr_nexthop *gr_nh, bool startup); void grout_nexthop_group_add(struct gr_nexthop *gr_nh, bool startup); diff --git a/frr/zebra_dplane_grout.c b/frr/zebra_dplane_grout.c index 2f901f678..e788cdc92 100644 --- a/frr/zebra_dplane_grout.c +++ b/frr/zebra_dplane_grout.c @@ -13,6 +13,8 @@ #include "log_grout.h" #include "rt_grout.h" +#include + #include #include #include @@ -90,6 +92,8 @@ static void zebra_grout_connect(struct event *); static void grout_sync(struct event *); static void grout_sync_ifaces(struct event *); static void grout_sync_addrs(struct event *); +static void grout_sync_lsps(struct event *); +static void grout_sync_routes(struct event *); static void grout_reconnect(struct event *); static void grout_reconnect_finish(void); static void grout_main_router_started(void); @@ -380,6 +384,53 @@ static void grout_sync_inject_marker(void) { ); } +static void grout_sync_lsps(struct event *e) { + struct gr_mpls_label_route_list_req req; + struct gr_mpls_label_route *route; + int ret; + + req = (struct gr_mpls_label_route_list_req) {.vrf_id = EVENT_VAL(e), .max_count = 0}; + + gr_log_info("vrf %u", EVENT_VAL(e)); + + gr_api_client_stream_foreach ( + route, ret, grout_ctx.sync_client, GR_MPLS_LABEL_ROUTE_LIST, sizeof(req), &req + ) { + grout_mpls_route_change(true, route, true); + } + if (ret < 0) { + gr_log_err("GR_MPLS_LABEL_ROUTE_LIST: %s", strerror(errno)); + event_add_timer( + zrouter.master, grout_reconnect, NULL, 1, &grout_ctx.dg_t_zebra_sync + ); + return; + } + + // Chain to next VRF's LSPs. + for (unsigned int i = EVENT_VAL(e) + 1; i < grout_ctx.max_ifaces; i++) { + if (bf_test_index(grout_ctx.sync_vrf, i)) { + event_add_event( + zrouter.master, grout_sync_lsps, NULL, i, &grout_ctx.dg_t_zebra_sync + ); + return; + } + } + + // All VRFs' LSPs done. Kick off Pass 4 (routes) from the first VRF. + for (unsigned int i = 0; i < grout_ctx.max_ifaces; i++) { + if (bf_test_index(grout_ctx.sync_vrf, i)) { + event_add_event( + zrouter.master, + grout_sync_routes, + NULL, + i, + &grout_ctx.dg_t_zebra_sync + ); + return; + } + } +} + static void grout_sync_routes(struct event *e) { struct gr_ip4_route_list_req r4_req = {.vrf_id = EVENT_VAL(e), .max_count = 0}; struct gr_ip4_route *r4; @@ -479,15 +530,11 @@ static void grout_sync_nh_groups(struct event *) { return; } - // Kick off routes starting from the first VRF. + // Kick off LSPs starting from the first VRF. for (unsigned int i = 0; i < grout_ctx.max_ifaces; i++) { if (bf_test_index(grout_ctx.sync_vrf, i)) { event_add_event( - zrouter.master, - grout_sync_routes, - NULL, - i, - &grout_ctx.dg_t_zebra_sync + zrouter.master, grout_sync_lsps, NULL, i, &grout_ctx.dg_t_zebra_sync ); return; } @@ -530,7 +577,7 @@ static void grout_sync_nhs(struct event *e) { } // Individual NHs done across all VRFs. Sync NH groups (global, not - // per-VRF) so that group member references resolve before routes. + // per-VRF) so that group member references resolve before LSPs and routes. event_add_event(zrouter.master, grout_sync_nh_groups, NULL, 0, &grout_ctx.dg_t_zebra_sync); } @@ -733,6 +780,8 @@ static void zebra_grout_connect(struct event *) { {.type = GR_EVENT_NEXTHOP_NEW, .suppress_self_events = true}, {.type = GR_EVENT_NEXTHOP_DELETE, .suppress_self_events = true}, {.type = GR_EVENT_NEXTHOP_UPDATE, .suppress_self_events = true}, + {.type = GR_EVENT_MPLS_ROUTE_ADD, .suppress_self_events = true}, + {.type = GR_EVENT_MPLS_ROUTE_DEL, .suppress_self_events = true}, }; if (grout_notif_subscribe(&grout_ctx.zebra_notifs, gr_evts, ARRAY_DIM(gr_evts)) < 0) { @@ -930,6 +979,12 @@ static void zebra_read_notifications(struct event *event) { case GR_EVENT_NEXTHOP_DELETE: grout_nexthop_change(new, PAYLOAD(gr_e), false); break; + case GR_EVENT_MPLS_ROUTE_ADD: + new = true; + // fallthrough + case GR_EVENT_MPLS_ROUTE_DEL: + grout_mpls_route_change(new, PAYLOAD(gr_e), false); + break; } free(gr_e); @@ -956,6 +1011,11 @@ static enum zebra_dplane_result zd_grout_process_update(struct zebra_dplane_ctx case DPLANE_OP_NH_DELETE: return grout_add_del_nexthop(ctx); + case DPLANE_OP_LSP_INSTALL: + case DPLANE_OP_LSP_UPDATE: + case DPLANE_OP_LSP_DELETE: + return grout_add_del_lsp(ctx); + case DPLANE_OP_MAC_INSTALL: case DPLANE_OP_MAC_DELETE: return grout_macfdb_update_ctx(ctx); From 64a9054d13b71c59e594c8916f1c26dc197ba135 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Fri, 2 Oct 2026 13:01:41 +0200 Subject: [PATCH 09/11] mpls: add control plane unit tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Unit tests for the MPLS control plane covering: - MPLS header encoding/decoding (mpls_hdr_set_label / mpls_hdr_get_label) - LFIB insert, lookup, delete, iteration, and boundary checks - MPLS nexthop equality (mpls_nh_equal via test wrapper) Wraps rte_zmalloc/rte_free so no DPDK EAL init is required. Moves cmocka_dep declaration before subdir('frr') in the top-level meson.build so the FRR subdir can reference it. Signed-off-by: Matej Mužila --- meson.build | 4 +- modules/mpls/control/meson.build | 10 ++ modules/mpls/control/mpls_test.c | 248 +++++++++++++++++++++++++++++++ modules/mpls/control/nexthop.c | 7 + 4 files changed, 268 insertions(+), 1 deletion(-) create mode 100644 modules/mpls/control/mpls_test.c diff --git a/meson.build b/meson.build index 842c2a032..8296da47b 100644 --- a/meson.build +++ b/meson.build @@ -162,6 +162,9 @@ subdir('main') subdir('modules') subdir('cli') subdir('api') + +cmocka_dep = dependency('cmocka', required: get_option('tests')) + subdir('frr') fs = import('fs') @@ -224,7 +227,6 @@ pkg.generate( install_dir: get_option('datadir') / 'pkgconfig', ) -cmocka_dep = dependency('cmocka', required: get_option('tests')) if cmocka_dep.found() foreach t : tests name = fs.replace_suffix(t['sources'].get(0), '').underscorify() diff --git a/modules/mpls/control/meson.build b/modules/mpls/control/meson.build index b404cc399..2463eeed6 100644 --- a/modules/mpls/control/meson.build +++ b/modules/mpls/control/meson.build @@ -8,3 +8,13 @@ src += files( 'resolve.c', ) inc += include_directories('.') + +tests += [ + { + 'sources': files('mpls_test.c', 'label_table.c', 'nexthop.c'), + 'link_args': [ + '-Wl,--wrap=rte_zmalloc', + '-Wl,--wrap=rte_free', + ], + }, +] diff --git a/modules/mpls/control/mpls_test.c b/modules/mpls/control/mpls_test.c new file mode 100644 index 000000000..58aa33604 --- /dev/null +++ b/modules/mpls/control/mpls_test.c @@ -0,0 +1,248 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "_cmocka.h" +#include "config.h" +#include "event.h" +#include "log.h" +#include "module.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include + +// Global variables declared extern in log.h; defined here for the test binary. +int gr_rte_log_type; +struct log_types log_types = STAILQ_HEAD_INITIALIZER(log_types); +struct gr_config gr_config; + +// Stubs for functions used by label_table.c and nexthop.c that are not under +// test. The prototypes are provided by the included headers above. +void module_register(struct module *) { } +void event_push(uint32_t, const void *) { } +void event_subscribe(uint32_t, event_sub_cb_t) { } +void nexthop_incref(struct nexthop *) { } +void nexthop_decref(struct nexthop *) { } +void nexthop_type_ops_register(gr_nh_type_t, const struct nexthop_type_ops *) { } +void vrf_fib_ops_register(addr_family_t, const struct vrf_fib_ops *) { } +struct nexthop *nexthop_new(const struct gr_nexthop_base *, const void *) { + return NULL; +} +struct nexthop *nexthop_lookup_l3(addr_family_t, uint16_t, uint16_t, const void *) { + return NULL; +} +void nexthop_iter(nh_iter_cb_t, void *) { } +struct iface *iface_from_id(uint16_t) { + return NULL; +} + +// rte_zmalloc wrapped to avoid DPDK EAL init in tests; forward declarations +// satisfy -Wmissing-prototypes before the definitions. +void *__wrap_rte_zmalloc(const char *, size_t size, unsigned /*align*/); +void *__wrap_rte_zmalloc(const char *, size_t size, unsigned /*align*/) { + return calloc(1, size); +} +void __wrap_rte_free(void *ptr); +void __wrap_rte_free(void *ptr) { + free(ptr); +} + +// Fake VRF iface: the flexible array trick lets iface_info_vrf(&fake_vrf.iface) +// point directly at the embedded vrf member. +static struct { + struct iface iface; + struct iface_info_vrf vrf; +} fake_vrf; + +struct iface *get_vrf_iface(uint16_t) { + return &fake_vrf.iface; +} + +static void test_label_encode_decode(void **) { + static const uint32_t labels[] = {0, 15, 16, 100, 0xFFFFF}; + for (size_t i = 0; i < sizeof(labels) / sizeof(labels[0]); i++) { + struct rte_mpls_hdr h = {}; + h.bs = 1; + h.tc = 7; + mpls_hdr_set_label(&h, labels[i]); + assert_int_equal(mpls_hdr_get_label(&h), labels[i]); + assert_int_equal(h.bs, 1); + assert_int_equal(h.tc, 7); + } +} + +static void test_label_split(void **) { + struct rte_mpls_hdr h = {}; + mpls_hdr_set_label(&h, 0x12345); + assert_int_equal(h.tag_lsb, 0x5); + assert_int_equal(mpls_hdr_get_label(&h), 0x12345); + + mpls_hdr_set_label(&h, 0); + assert_int_equal(h.tag_lsb, 0); + assert_int_equal(mpls_hdr_get_label(&h), 0); +} + +static int lfib_setup(void **) { + size_t sz; + + memset(&fake_vrf, 0, sizeof(fake_vrf)); + fake_vrf.iface.type = GR_IFACE_TYPE_VRF; + sz = (GR_MPLS_LABEL_MAX + 1) * sizeof(struct nexthop *); + iface_info_vrf(&fake_vrf.iface)->fib_mpls = calloc(1, sz); + return 0; +} + +static int lfib_teardown(void **) { + free(iface_info_vrf(&fake_vrf.iface)->fib_mpls); + iface_info_vrf(&fake_vrf.iface)->fib_mpls = NULL; + return 0; +} + +static void test_lfib_insert_lookup(void **) { + struct nexthop nh = {}; + nh.origin = GR_NH_ORIGIN_ZSTATIC; + + assert_int_equal(mpls_rib_insert(1, 100, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_ptr_equal(mpls_fib_lookup(1, 100), &nh); + assert_null(mpls_fib_lookup(1, 101)); +} + +static void test_lfib_insert_duplicate(void **) { + struct nexthop nh = {}; + nh.origin = GR_NH_ORIGIN_ZSTATIC; + + assert_int_equal(mpls_rib_insert(1, 200, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_int_equal(mpls_rib_insert(1, 200, &nh, GR_NH_ORIGIN_ZSTATIC, true), 0); + assert_int_equal(mpls_rib_insert(1, 200, &nh, GR_NH_ORIGIN_ZSTATIC, false), -EEXIST); +} + +static void test_lfib_delete(void **) { + struct nexthop nh = {}; + nh.origin = GR_NH_ORIGIN_ZSTATIC; + + assert_int_equal(mpls_rib_insert(1, 300, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_int_equal(mpls_rib_delete(1, 300, false), 0); + assert_null(mpls_fib_lookup(1, 300)); + assert_int_equal(mpls_rib_delete(1, 300, true), 0); + assert_int_equal(mpls_rib_delete(1, 300, false), -ENOENT); +} + +static int g_iter_count; +static int count_cb(uint16_t, uint32_t, const struct nexthop *, void *) { + g_iter_count++; + return 0; +} + +static void test_lfib_iter(void **) { + struct nexthop nh = {}; + nh.origin = GR_NH_ORIGIN_ZSTATIC; + + assert_int_equal(mpls_rib_insert(1, 400, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_int_equal(mpls_rib_insert(1, 401, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_int_equal(mpls_rib_insert(1, 402, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + + g_iter_count = 0; + assert_int_equal(mpls_rib_iter(1, count_cb, NULL), 0); + assert_int_equal(g_iter_count, 3); +} + +static void test_lfib_label_invalid(void **) { + struct nexthop nh = {}; + uint32_t bad; + + bad = GR_MPLS_LABEL_MAX + 1; + + assert_int_equal(mpls_rib_insert(1, bad, &nh, GR_NH_ORIGIN_ZSTATIC, false), -EINVAL); + assert_null(mpls_fib_lookup(1, bad)); + assert_int_equal(mpls_rib_delete(1, bad, false), -EINVAL); +} + +extern bool mpls_nh_equal_test(const struct nexthop *, const struct nexthop *); + +static void test_nh_equal_identical(void **) { + struct nexthop a = {}, b = {}; + a.type = GR_NH_T_MPLS; + b.type = GR_NH_T_MPLS; + struct nexthop_info_mpls *ma = nexthop_info_mpls(&a); + struct nexthop_info_mpls *mb = nexthop_info_mpls(&b); + + ma->n_labels = 1; + ma->labels[0] = 100; + ma->payload_af = GR_AF_UNSPEC; + ma->via_nh = NULL; + memcpy(mb, ma, sizeof(*mb)); + + assert_true(mpls_nh_equal_test(&a, &b)); +} + +static void test_nh_equal_different_n_labels(void **) { + struct nexthop a = {}, b = {}; + a.type = GR_NH_T_MPLS; + b.type = GR_NH_T_MPLS; + struct nexthop_info_mpls *ma = nexthop_info_mpls(&a); + struct nexthop_info_mpls *mb = nexthop_info_mpls(&b); + + ma->n_labels = 1; + ma->labels[0] = 100; + mb->n_labels = 2; + mb->labels[0] = 100; + mb->labels[1] = 200; + + assert_false(mpls_nh_equal_test(&a, &b)); +} + +static void test_nh_equal_different_label(void **) { + struct nexthop a = {}, b = {}; + a.type = GR_NH_T_MPLS; + b.type = GR_NH_T_MPLS; + struct nexthop_info_mpls *ma = nexthop_info_mpls(&a); + struct nexthop_info_mpls *mb = nexthop_info_mpls(&b); + + ma->n_labels = 1; + ma->labels[0] = 100; + mb->n_labels = 1; + mb->labels[0] = 200; + + assert_false(mpls_nh_equal_test(&a, &b)); +} + +static void test_nh_equal_different_payload_af(void **) { + struct nexthop a = {}, b = {}; + a.type = GR_NH_T_MPLS; + b.type = GR_NH_T_MPLS; + struct nexthop_info_mpls *ma = nexthop_info_mpls(&a); + struct nexthop_info_mpls *mb = nexthop_info_mpls(&b); + + ma->n_labels = 1; + ma->labels[0] = 100; + ma->payload_af = GR_AF_IP4; + mb->n_labels = 1; + mb->labels[0] = 100; + mb->payload_af = GR_AF_IP6; + + assert_false(mpls_nh_equal_test(&a, &b)); +} + +int main(void) { + const struct CMUnitTest tests[] = { + // Group A: header encoding + cmocka_unit_test(test_label_encode_decode), + cmocka_unit_test(test_label_split), + // Group B: LFIB operations + cmocka_unit_test_setup_teardown(test_lfib_insert_lookup, lfib_setup, lfib_teardown), + cmocka_unit_test_setup_teardown( + test_lfib_insert_duplicate, lfib_setup, lfib_teardown + ), + cmocka_unit_test_setup_teardown(test_lfib_delete, lfib_setup, lfib_teardown), + cmocka_unit_test_setup_teardown(test_lfib_iter, lfib_setup, lfib_teardown), + cmocka_unit_test_setup_teardown(test_lfib_label_invalid, lfib_setup, lfib_teardown), + // Group C: nexthop equality + cmocka_unit_test(test_nh_equal_identical), + cmocka_unit_test(test_nh_equal_different_n_labels), + cmocka_unit_test(test_nh_equal_different_label), + cmocka_unit_test(test_nh_equal_different_payload_af), + }; + return cmocka_run_group_tests(tests, NULL, NULL); +} diff --git a/modules/mpls/control/nexthop.c b/modules/mpls/control/nexthop.c index f91124f90..35c9ad1ec 100644 --- a/modules/mpls/control/nexthop.c +++ b/modules/mpls/control/nexthop.c @@ -141,3 +141,10 @@ static struct nexthop_type_ops mpls_nh_ops = { RTE_INIT(mpls_nexthop_init) { nexthop_type_ops_register(GR_NH_T_MPLS, &mpls_nh_ops); } + +#ifdef __GROUT_UNIT_TEST__ +bool mpls_nh_equal_test(const struct nexthop *a, const struct nexthop *b); +bool mpls_nh_equal_test(const struct nexthop *a, const struct nexthop *b) { + return mpls_nh_equal(a, b); +} +#endif From 340652bddbca0df0e20e8a1386979788e9eb0c0e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Fri, 2 Oct 2026 13:01:54 +0200 Subject: [PATCH 10/11] frr: add unit tests for MPLS helper functions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Unit tests for three functions in rt_grout.c that can be exercised without a live grout daemon or FRR zebra process: - lsptype2origin: verify all ZEBRA_LSP_* → GR_NH_ORIGIN_* mappings - nh_has_mpls_labels: NULL/zero/implicit-null/real-label cases - grout_fill_mpls_nh: IPv4/IPv6 label stacks, too-many-labels, unsupported type The three functions are exposed as non-static under __GROUT_UNIT_TEST__ via the new TESTABLE_STATIC macro. Functions that require a live daemon (grout_add_del_lsp, etc.) are guarded with #ifndef __GROUT_UNIT_TEST__. The test executable links against the installed libfrr.so directly since the dplane_grout plugin normally inherits FRR symbols from the parent process at runtime rather than linking them explicitly. Signed-off-by: Matej Mužila --- frr/meson.build | 19 ++++ frr/mpls_frr_test.c | 235 ++++++++++++++++++++++++++++++++++++++++++++ frr/rt_grout.c | 21 +++- frr/rt_grout.h | 13 +++ 4 files changed, 285 insertions(+), 3 deletions(-) create mode 100644 frr/mpls_frr_test.c diff --git a/frr/meson.build b/frr/meson.build index b8311aede..3714d6d98 100644 --- a/frr/meson.build +++ b/frr/meson.build @@ -61,6 +61,25 @@ if install_build_flag ) endif +if cmocka_dep.found() + # FRR is built as an external subproject and only provides include dirs + # (the dplane_grout plugin inherits FRR symbols from the parent process at + # runtime). For a standalone test binary we must link libfrr explicitly. + frr_prefix = frr_dep.get_variable('prefix') + frr_libdir = frr_prefix / 'lib' + mpls_frr_test_exe = executable( + 'mpls_frr_test', + files('mpls_frr_test.c', 'rt_grout.c') + grout_header, + c_args: frr_c_args + ['-D__GROUT_UNIT_TEST__'], + include_directories: api_inc + include_directories('.'), + dependencies: [frr_dep, cmocka_dep], + link_args: ['-L' + frr_libdir, '-lfrr', '-Wl,-rpath,' + frr_libdir], + install: false, + override_options: ['b_sanitize=none'], + ) + test('mpls_frr_test', mpls_frr_test_exe, suite: 'unit') +endif + systemd_dep = dependency('systemd', required: false) if systemd_dep.found() systemd_system_unit_dir = systemd_dep.get_variable( diff --git a/frr/mpls_frr_test.c b/frr/mpls_frr_test.c new file mode 100644 index 000000000..930656aa2 --- /dev/null +++ b/frr/mpls_frr_test.c @@ -0,0 +1,235 @@ +// SPDX-License-Identifier: GPL-2.0-or-later +// Copyright (c) 2025 Matej Muzila + +#include "rt_grout.h" + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +// Forward declarations matching if_map.h so -Wmissing-prototypes is satisfied. +// These stubs replace if_map.c which is not linked into the test binary. +uint16_t ifindex_frr_to_grout(ifindex_t); +uint16_t vrf_frr_to_grout(vrf_id_t); + +uint16_t ifindex_frr_to_grout(ifindex_t) { + return 0; +} +uint16_t vrf_frr_to_grout(vrf_id_t) { + return 0; +} + +static void test_lsptype2origin(void **) { + assert_int_equal(lsptype2origin(ZEBRA_LSP_STATIC), GR_NH_ORIGIN_ZSTATIC); + assert_int_equal(lsptype2origin(ZEBRA_LSP_LDP), GR_NH_ORIGIN_LDP); + assert_int_equal(lsptype2origin(ZEBRA_LSP_BGP), GR_NH_ORIGIN_BGP); + assert_int_equal(lsptype2origin(ZEBRA_LSP_OSPF_SR), GR_NH_ORIGIN_OSPF); + assert_int_equal(lsptype2origin(ZEBRA_LSP_ISIS_SR), GR_NH_ORIGIN_ISIS); + assert_int_equal(lsptype2origin(ZEBRA_LSP_SHARP), GR_NH_ORIGIN_SHARP); + assert_int_equal(lsptype2origin(ZEBRA_LSP_SRTE), GR_NH_ORIGIN_SRTE); + assert_int_equal(lsptype2origin(ZEBRA_LSP_NONE), GR_NH_ORIGIN_ZEBRA); + assert_int_equal(lsptype2origin(ZEBRA_LSP_EVPN), GR_NH_ORIGIN_ZEBRA); +} + +static void test_nh_has_mpls_labels_null_label(void **) { + struct nexthop nh = {}; + + assert_false(nh_has_mpls_labels(&nh)); +} + +static void test_nh_has_mpls_labels_zero_count(void **) { + struct mpls_label_stack nhl = {.num_labels = 0}; + struct nexthop nh = {}; + + nh.nh_label = &nhl; + assert_false(nh_has_mpls_labels(&nh)); +} + +static void test_nh_has_mpls_labels_implicit_null(void **) { + struct mpls_label_stack *nhl; + struct nexthop nh = {}; + + nhl = malloc(sizeof(*nhl) + sizeof(mpls_label_t)); + + assert_non_null(nhl); + nhl->num_labels = 1; + nhl->label[0] = MPLS_LABEL_IMPLICIT_NULL; + nh.nh_label = nhl; + assert_false(nh_has_mpls_labels(&nh)); + free(nhl); +} + +static void test_nh_has_mpls_labels_real_label(void **) { + struct mpls_label_stack *nhl; + struct nexthop nh = {}; + + nhl = malloc(sizeof(*nhl) + sizeof(mpls_label_t)); + + assert_non_null(nhl); + nhl->num_labels = 1; + nhl->label[0] = 100; + nh.nh_label = nhl; + assert_true(nh_has_mpls_labels(&nh)); + free(nhl); +} + +static void test_nh_has_mpls_labels_two_labels(void **) { + struct mpls_label_stack *nhl; + struct nexthop nh = {}; + + nhl = malloc(sizeof(*nhl) + 2 * sizeof(mpls_label_t)); + + assert_non_null(nhl); + nhl->num_labels = 2; + nhl->label[0] = 100; + nhl->label[1] = 200; + nh.nh_label = nhl; + assert_true(nh_has_mpls_labels(&nh)); + free(nhl); +} + +static void test_fill_mpls_nh_ipv4_one_label(void **) { + struct gr_nexthop_info_mpls *mpls; + struct mpls_label_stack *nhl; + struct gr_nh_add_req *req; + struct nexthop nh = {}; + size_t len; + + len = sizeof(struct gr_nh_add_req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + mpls = (struct gr_nexthop_info_mpls *)req->nh.info; + nhl = malloc(sizeof(*nhl) + sizeof(mpls_label_t)); + + assert_non_null(req); + + nh.type = NEXTHOP_TYPE_IPV4_IFINDEX; + nh.gate.ipv4.s_addr = htonl(0xAC100102); + + assert_non_null(nhl); + nhl->num_labels = 1; + nhl->label[0] = 100; + nh.nh_label = nhl; + + assert_int_equal(grout_fill_mpls_nh(req, 42, GR_NH_ORIGIN_ZSTATIC, &nh), 0); + + assert_int_equal(mpls->via.af, GR_AF_IP4); + assert_int_equal(mpls->n_labels, 1); + assert_int_equal(mpls->labels[0], 100); + assert_int_equal(mpls->ttl, 0); + assert_int_equal(mpls->payload_af, GR_AF_UNSPEC); + assert_true(req->exist_ok); + assert_int_equal(req->nh.nh_id, 42); + assert_int_equal(req->nh.origin, GR_NH_ORIGIN_ZSTATIC); + assert_int_equal(req->nh.type, GR_NH_T_MPLS); + + free(nhl); + free(req); +} + +static void test_fill_mpls_nh_ipv6_two_labels(void **) { + struct gr_nexthop_info_mpls *mpls; + struct mpls_label_stack *nhl; + struct gr_nh_add_req *req; + struct nexthop nh = {}; + size_t len; + + len = sizeof(struct gr_nh_add_req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + mpls = (struct gr_nexthop_info_mpls *)req->nh.info; + nhl = malloc(sizeof(*nhl) + 2 * sizeof(mpls_label_t)); + + assert_non_null(req); + + nh.type = NEXTHOP_TYPE_IPV6_IFINDEX; + nh.gate.ipv6.s6_addr[0] = 0xfe; + nh.gate.ipv6.s6_addr[1] = 0x80; + nh.gate.ipv6.s6_addr[15] = 0x01; + + assert_non_null(nhl); + nhl->num_labels = 2; + nhl->label[0] = 200; + nhl->label[1] = 100; + nh.nh_label = nhl; + + assert_int_equal(grout_fill_mpls_nh(req, 43, GR_NH_ORIGIN_BGP, &nh), 0); + + assert_int_equal(mpls->via.af, GR_AF_IP6); + assert_int_equal(mpls->n_labels, 2); + assert_int_equal(mpls->labels[0], 200); + assert_int_equal(mpls->labels[1], 100); + + free(nhl); + free(req); +} + +static void test_fill_mpls_nh_too_many_labels(void **) { + struct mpls_label_stack *nhl; + struct gr_nh_add_req *req; + struct nexthop nh = {}; + size_t len; + uint8_t n; + + n = GR_MPLS_MAX_LABELS + 1; + len = sizeof(struct gr_nh_add_req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + nhl = malloc(sizeof(*nhl) + n * sizeof(mpls_label_t)); + + assert_non_null(req); + + nh.type = NEXTHOP_TYPE_IPV4; + nh.gate.ipv4.s_addr = htonl(0xAC100102); + + assert_non_null(nhl); + nhl->num_labels = n; + for (uint8_t i = 0; i < n; i++) + nhl->label[i] = 100 + i; + nh.nh_label = nhl; + + assert_int_equal(grout_fill_mpls_nh(req, 44, GR_NH_ORIGIN_ZSTATIC, &nh), -1); + + free(nhl); + free(req); +} + +static void test_fill_mpls_nh_unsupported_type(void **) { + struct gr_nh_add_req *req; + struct nexthop nh = {}; + size_t len; + + len = sizeof(struct gr_nh_add_req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + + assert_non_null(req); + + nh.type = NEXTHOP_TYPE_BLACKHOLE; + + assert_int_equal(grout_fill_mpls_nh(req, 45, GR_NH_ORIGIN_ZSTATIC, &nh), -1); + + free(req); +} + +int main(void) { + const struct CMUnitTest tests[] = { + cmocka_unit_test(test_lsptype2origin), + cmocka_unit_test(test_nh_has_mpls_labels_null_label), + cmocka_unit_test(test_nh_has_mpls_labels_zero_count), + cmocka_unit_test(test_nh_has_mpls_labels_implicit_null), + cmocka_unit_test(test_nh_has_mpls_labels_real_label), + cmocka_unit_test(test_nh_has_mpls_labels_two_labels), + cmocka_unit_test(test_fill_mpls_nh_ipv4_one_label), + cmocka_unit_test(test_fill_mpls_nh_ipv6_two_labels), + cmocka_unit_test(test_fill_mpls_nh_too_many_labels), + cmocka_unit_test(test_fill_mpls_nh_unsupported_type), + }; + return cmocka_run_group_tests(tests, NULL, NULL); +} diff --git a/frr/rt_grout.c b/frr/rt_grout.c index 6be466941..b125dfb96 100644 --- a/frr/rt_grout.c +++ b/frr/rt_grout.c @@ -28,6 +28,16 @@ // (<= 1 << 17) and FRR's proto-NHG range (< ZEBRA_NHG_PROTO_UPPER, ~250M). #define GROUT_MPLS_NH_ID(label) (0x80000000u | (uint32_t)(label)) +// Under __GROUT_UNIT_TEST__, expose the three MPLS helper functions so that +// mpls_frr_test.c can call them directly without the full plugin link context. +#ifdef __GROUT_UNIT_TEST__ +#define TESTABLE_STATIC +#else +#define TESTABLE_STATIC static +#endif + +#ifndef __GROUT_UNIT_TEST__ + static inline bool is_selfroute(gr_nh_origin_t origin) { switch (origin) { case GR_NH_ORIGIN_ZEBRA: @@ -650,7 +660,9 @@ enum zebra_dplane_result grout_add_del_route(struct zebra_dplane_ctx *ctx) { return ZEBRA_DPLANE_REQUEST_SUCCESS; } -static inline gr_nh_origin_t lsptype2origin(enum lsp_types_t type) { +#endif /* !__GROUT_UNIT_TEST__ */ + +TESTABLE_STATIC inline gr_nh_origin_t lsptype2origin(enum lsp_types_t type) { switch (type) { case ZEBRA_LSP_STATIC: return GR_NH_ORIGIN_ZSTATIC; @@ -676,7 +688,7 @@ static inline gr_nh_origin_t lsptype2origin(enum lsp_types_t type) { // A lone implicit-null is penultimate hop popping, which grout // represents as an MPLS nexthop with zero output labels, so it does not count // as "having labels" here. -static bool nh_has_mpls_labels(const struct nexthop *nh) { +TESTABLE_STATIC bool nh_has_mpls_labels(const struct nexthop *nh) { const struct mpls_label_stack *nhl = nh->nh_label; return nhl != NULL && nhl->num_labels > 0 @@ -686,7 +698,7 @@ static bool nh_has_mpls_labels(const struct nexthop *nh) { // Populate an MPLS nexthop add request from a FRR nexthop carrying output // labels. Returns 0 on success, -1 if the nexthop cannot be represented in // grout. The caller must allocate req with room for a gr_nexthop_info_mpls. -static int grout_fill_mpls_nh( +TESTABLE_STATIC int grout_fill_mpls_nh( struct gr_nh_add_req *req, uint32_t nh_id, gr_nh_origin_t origin, @@ -740,6 +752,8 @@ static int grout_fill_mpls_nh( return 0; } +#ifndef __GROUT_UNIT_TEST__ + enum zebra_dplane_result grout_add_del_lsp(struct zebra_dplane_ctx *ctx) { struct gr_mpls_label_route_del_req del; struct gr_mpls_label_route_add_req add; @@ -1649,3 +1663,4 @@ enum zebra_dplane_result grout_neigh_read_ctx(struct zebra_dplane_ctx *ctx) { return ZEBRA_DPLANE_REQUEST_SUCCESS; } #endif +#endif /* !__GROUT_UNIT_TEST__ */ diff --git a/frr/rt_grout.h b/frr/rt_grout.h index aafdde20c..f471806f1 100644 --- a/frr/rt_grout.h +++ b/frr/rt_grout.h @@ -26,3 +26,16 @@ enum zebra_dplane_result grout_neigh_update_ctx(struct zebra_dplane_ctx *ctx); enum zebra_dplane_result grout_vxlan_flood_update_ctx(struct zebra_dplane_ctx *ctx); enum zebra_dplane_result grout_fdb_read_ctx(struct zebra_dplane_ctx *ctx); enum zebra_dplane_result grout_neigh_read_ctx(struct zebra_dplane_ctx *ctx); + +#ifdef __GROUT_UNIT_TEST__ +#include +#include +gr_nh_origin_t lsptype2origin(enum lsp_types_t type); +bool nh_has_mpls_labels(const struct nexthop *nh); +int grout_fill_mpls_nh( + struct gr_nh_add_req *req, + uint32_t nh_id, + gr_nh_origin_t origin, + const struct nexthop *nh +); +#endif From b050cbf67a172dfdd41a762057d93fe1de5049a9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Matej=20Mu=C5=BEila?= Date: Fri, 2 Oct 2026 13:02:19 +0200 Subject: [PATCH 11/11] mpls: add FRR smoke tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit End-to-end tests for the FRR → dplane_grout → grout MPLS path: - staticd: static LSP (PHP and label swap) via FRR staticd - OSPF-SR: automatic LSP installation via OSPF Segment Routing - IS-IS SR: automatic LSP installation via IS-IS Segment Routing Signed-off-by: Matej Mužila --- smoke/isis_sr_frr_test.sh | 121 ++++++++++++++++++++++++++++++++++ smoke/mpls_static_frr_test.sh | 83 +++++++++++++++++++++++ smoke/ospf_sr_frr_test.sh | 120 +++++++++++++++++++++++++++++++++ 3 files changed, 324 insertions(+) create mode 100755 smoke/isis_sr_frr_test.sh create mode 100755 smoke/mpls_static_frr_test.sh create mode 100755 smoke/ospf_sr_frr_test.sh diff --git a/smoke/isis_sr_frr_test.sh b/smoke/isis_sr_frr_test.sh new file mode 100755 index 000000000..99e1484f5 --- /dev/null +++ b/smoke/isis_sr_frr_test.sh @@ -0,0 +1,121 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +# Verify that IS-IS Segment Routing automatically installs LSPs in grout via +# dplane_grout. isisd computes the penultimate-hop LSP for the peer's node SID +# and sends it to zebra → dplane_grout → grout LFIB. +# +# Both grout and the peer use SRGB [16000, 23999]: +# grout prefix SID index 0 → label 16000 (explicit-null toward grout) +# peer prefix SID index 1 → label 16001 (PHP at grout for peer) +# +# IS-IS SR installs at grout: in=16001, PHP (implicit-null), via 172.16.0.2 +# +# .--------------------. +# | netns "isis-peer" | +# .--------..------------. | .-------. | +# | zebra || grout | | | isisd | | +# '--------'| | | '-------' | +# .-------. | .------------. .------------. .-------. | +# | isisd | | | p0 | net_tap | x-p0 | | zebra | | +# '-------' | | +-------------+ | '-------' | +# .------. | 172.16.0.1 | | 172.16.0.2 |.----------. | +# | main | '------------' '------------'| lo | | +# '------' | | | | | +# | ping <------------------------------------> | 16.0.0.1 | | +# | | | '----------' | +# '-----------' '--------------------' + +. $(dirname $0)/_init_frr.sh + +create_interface p0 +set_ip_address p0 172.16.0.1/24 + +# Configure grout's FRR: IS-IS with Segment Routing +vtysh <<-EOF +configure terminal +! +ip router-id 172.16.0.1 +! +interface lo + ip address 17.0.0.1/32 +exit +! +interface p0 + ip router isis smoke + isis network point-to-point +exit +! +router isis smoke + net 49.0000.0000.0001.00 + is-type level-2-only + redistribute ipv4 connected level-2 + segment-routing on + segment-routing global-block 16000 23999 + segment-routing node-msd 8 + segment-routing prefix 17.0.0.1/32 index 0 explicit-null +exit +! +EOF + +start_frr isis-peer 0 +ip link set x-p0 netns isis-peer + +# Enable kernel MPLS in the peer netns so that FRR zebra can install the +# pop rule for its own node SID (label 16001) when IS-IS SR comes up. +ip netns exec isis-peer sysctl -wq net.mpls.platform_labels=1000 +ip netns exec isis-peer sysctl -wq net.mpls.conf.x-p0.input=1 + +vtysh -N isis-peer <<-EOF +configure terminal +! +ip router-id 172.16.0.2 +! +interface lo + ip address 16.0.0.1/32 +exit +! +interface x-p0 + ip address 172.16.0.2/24 + ip router isis smoke + isis network point-to-point +exit +! +router isis smoke + net 49.0000.0000.0002.00 + is-type level-2-only + redistribute ipv4 connected level-2 + segment-routing on + segment-routing global-block 16000 23999 + segment-routing node-msd 8 + segment-routing prefix 16.0.0.1/32 index 1 +exit +! +EOF + +# Wait for IS-IS adjacency +attempts=60 +while ! vtysh -c 'show isis neighbor json' | jq -e '.areas[0].circuits[0].state == "Up"'; do + sleep 1 + if [ "$attempts" -le 0 ]; then + fail "IS-IS failed to connect to neighbor." + fi + attempts=$((attempts - 1)) +done + +# Wait for IS-IS route exchange +attempts=90 +while ! vtysh -c 'show ip route isis json' | jq -e '."16.0.0.1/32"'; do + sleep 1 + if [ "$attempts" -le 0 ]; then + fail "IS-IS failed to get routes." + fi + attempts=$((attempts - 1)) +done + +# IS-IS SR installs LSP for peer's node SID (label 16001) in grout LFIB +wait_event -t 60 'mpls route add: vrf=main label=16001 .* origin=isis' + +# IP connectivity via IS-IS learned route +grcli ping 16.0.0.1 count 3 delay 10 diff --git a/smoke/mpls_static_frr_test.sh b/smoke/mpls_static_frr_test.sh new file mode 100755 index 000000000..2260f0374 --- /dev/null +++ b/smoke/mpls_static_frr_test.sh @@ -0,0 +1,83 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +# Verify the FRR staticd → dplane_grout → grout MPLS LSP path for two cases: +# 1. PHP (implicit-null): in-label 100, grout pops and forwards to n1 +# 2. Label swap: in-label 100 → out-label 200, grout swaps and n1 pops +# +# .-------..--------------. +# | zebra || grout | +# '-------'| | +# .-------.|.-----..------.. +# |staticd||| p0 || p1 || +# '-------'|| 172|| 172 || +# ||16.0.1|16.1.1|| +# .----------------. .-----------. +# | n0 | | n1 | +# | 172.16.0.2/24 | |172.16.1.2 | +# | push MPLS 100 | |10.0.1.1/lo| +# '----------------' '-----------' +# +# PHP flow: +# n0 --[label 100]--> grout --[pop]--> n1 (10.0.1.1) +# Swap flow: +# n0 --[label 100]--> grout --[swap→200]--> n1 --[kernel pop 200]--> lo + +. $(dirname $0)/_init_frr.sh + +create_interface p0 +create_interface p1 + +set_ip_address p0 172.16.0.1/24 +set_ip_address p1 172.16.1.1/24 + +netns_add n0 +netns_add n1 +move_to_netns x-p0 n0 +move_to_netns x-p1 n1 + +ip -n n0 addr add 172.16.0.2/24 dev x-p0 +ip -n n1 addr add 172.16.1.2/24 dev x-p1 +ip -n n1 addr add 10.0.1.1/32 dev lo + +# n1: reply path for pings originating from n0 (172.16.0.2) +ip -n n1 route add 172.16.0.0/24 via 172.16.1.1 + +# grout: IP route to 10.0.1.0/24 so that after PHP it can forward to n1 +set_ip_route 10.0.1.0/24 172.16.1.2 + +# Install MPLS LSP via FRR staticd: in-label 100, PHP (implicit-null) via 172.16.1.2 +_apply_frr_config 0 \ + "mpls route add: vrf=main label=100 .* origin=zebra_static" \ + "mpls lsp 100 172.16.1.2 implicit-null" + +# n0: push MPLS label 100 for traffic to 10.0.1.0/24 via grout +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 + +# n0 sends MPLS-labeled packet, grout pops label 100 (PHP) and forwards to n1 +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 + +# Remove LSP and verify removal event +_apply_frr_config 0 \ + "mpls route del: vrf=main label=100 .* origin=zebra_static" \ + "no mpls lsp 100 172.16.1.2 implicit-null" + +# Label swap: in-label 100 → out-label 200 via 172.16.1.2 +# n1 needs kernel MPLS to accept and pop label 200 on ingress +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 -f mpls route add 200 dev lo + +_apply_frr_config 0 \ + "mpls route add: vrf=main label=100 .* origin=zebra_static" \ + "mpls lsp 100 172.16.1.2 200" + +# n0 sends MPLS label 100, grout swaps to 200, n1 pops and delivers to lo +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 + +# Remove swap LSP +_apply_frr_config 0 \ + "mpls route del: vrf=main label=100 .* origin=zebra_static" \ + "no mpls lsp 100 172.16.1.2 200" diff --git a/smoke/ospf_sr_frr_test.sh b/smoke/ospf_sr_frr_test.sh new file mode 100755 index 000000000..dcae4bf9b --- /dev/null +++ b/smoke/ospf_sr_frr_test.sh @@ -0,0 +1,120 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +# Verify that OSPF Segment Routing automatically installs LSPs in grout via +# dplane_grout. ospfd computes the penultimate-hop LSP for the peer's node SID +# and sends it to zebra → dplane_grout → grout LFIB. +# +# Both grout and the peer use SRGB [16000, 23999]: +# grout prefix SID index 0 → label 16000 (or 15000 depending on ospfd timing) +# peer prefix SID index 1 → label 16001 (or 15001 depending on ospfd timing) +# +# OSPF-SR installs at grout: in=peer's SID, PHP (implicit-null), via 172.16.0.2 +# +# .--------------------. +# | netns "ospf-peer" | +# .--------..------------. | .-------. | +# | zebra || grout | | | ospfd | | +# '--------'| | | '-------' | +# .-------. | .------------. .------------. .-------. | +# | ospfd | | | p0 | net_tap | x-p0 | | zebra | | +# '-------' | | +-------------+ | '-------' | +# .------. | 172.16.0.1 | | 172.16.0.2 |.----------. | +# | main | '------------' '------------'| lo | | +# '------' | | | | | +# | ping <------------------------------------> | 16.0.0.1 | | +# | | | '----------' | +# '-----------' '--------------------' + +. $(dirname $0)/_init_frr.sh + +create_interface p0 +set_ip_address p0 172.16.0.1/24 + +start_frr ospf-peer 0 +move_to_netns x-p0 ospf-peer + +# Enable kernel MPLS in the peer netns so that FRR zebra can install the +# pop rule for its own node SID (label 16001) when OSPF-SR comes up. +ip netns exec ospf-peer sysctl -wq net.mpls.platform_labels=17000 +ip netns exec ospf-peer sysctl -wq net.mpls.conf.x-p0.input=1 + +# Configure grout's FRR: OSPF with Segment Routing +vtysh <<-EOF +configure terminal +ip router-id 172.16.0.1 +! +interface lo + ip address 17.0.0.1/32 +exit +! +interface p0 + ip ospf hello-interval 1 + ip ospf network point-to-point +exit +! +router ospf + ospf router-id 172.16.0.1 + network 172.16.0.0/24 area 0 + network 17.0.0.1/32 area 0 + router-info area + segment-routing global-block 16000 23999 + segment-routing on + segment-routing node-msd 8 + segment-routing prefix 17.0.0.1/32 index 0 +exit +! +EOF + +vtysh -N ospf-peer <<-EOF +configure terminal +ip router-id 172.16.0.2 +! +interface lo + ip address 16.0.0.1/32 +exit +! +interface x-p0 + ip address 172.16.0.2/24 + ip ospf hello-interval 1 + ip ospf network point-to-point +exit +! +router ospf + ospf router-id 172.16.0.2 + network 172.16.0.0/24 area 0 + network 16.0.0.1/32 area 0 + router-info area + segment-routing global-block 16000 23999 + segment-routing on + segment-routing node-msd 8 + segment-routing prefix 16.0.0.1/32 index 1 +exit +! +EOF + +attempts=60 +while ! vtysh -c 'show ip ospf neighbor json' | jq -e '.neighbors."172.16.0.2"[0].converged == "Full"'; do + sleep 1 + if [ "$attempts" -le 0 ]; then + fail "OSPF failed to connect to neighbor." + fi + attempts=$((attempts - 1)) +done + +# Wait for OSPF route exchange +attempts=30 +while ! vtysh -c 'show ip route ospf json' | jq -e '."16.0.0.1/32"'; do + sleep 1 + if [ "$attempts" -le 0 ]; then + fail "OSPF failed to get routes." + fi + attempts=$((attempts - 1)) +done + +# OSPF-SR installs LSP for peer's node SID in grout LFIB +wait_event -t 60 'mpls route add: vrf=main label=1[56][0-9][0-9][0-9] .* origin=ospf' + +# IP route is installed by dplane_grout synchronously with the RIB update above +grcli ping 16.0.0.1 count 1