diff --git a/api/gr_net_types.h b/api/gr_net_types.h index efaf0c830..079bb9566 100644 --- a/api/gr_net_types.h +++ b/api/gr_net_types.h @@ -27,6 +27,7 @@ typedef enum : uint8_t { GR_AF_UNSPEC = AF_UNSPEC, GR_AF_IP4 = AF_INET, GR_AF_IP6 = AF_INET6, + GR_AF_MPLS = AF_MPLS, } addr_family_t; // Convert address family enum to string representation. @@ -38,6 +39,8 @@ static inline const char *gr_af_name(addr_family_t af) { return "ipv4"; case GR_AF_IP6: return "ipv6"; + case GR_AF_MPLS: + return "mpls"; } return "?"; } @@ -48,6 +51,7 @@ static inline bool gr_af_valid(addr_family_t af) { case GR_AF_UNSPEC: case GR_AF_IP4: case GR_AF_IP6: + case GR_AF_MPLS: return true; } return false; diff --git a/docs/graph.svg b/docs/graph.svg index a3e455385..6051a923d 100644 --- a/docs/graph.svg +++ b/docs/graph.svg @@ -4,957 +4,1101 @@ - - - + + + bond_output - -bond_output + +bond_output port_output - -port_output + +port_output bond_output->port_output - - + + iface_input - -iface_input + +iface_input xconnect - -xconnect + +xconnect - + iface_input->xconnect - - + + eth_input - -eth_input + +eth_input - + iface_input->eth_input - - + + bridge_input - -bridge_input + +bridge_input - + iface_input->bridge_input - - + + iface_output - -iface_output + +iface_output - + iface_output->bond_output - - + + - + iface_output->port_output - - + + - + iface_output->bridge_input - - + + - + vxlan_output - -vxlan_output + +vxlan_output - + iface_output->vxlan_output - - + + port_tx - -port_tx + +port_tx - + port_output->port_tx - - + + port_rx - -port_rx + +port_rx - + port_rx->iface_input - - + + - + xconnect->port_output - - + + lacp_input - -lacp_input + +lacp_input eth_input->lacp_input - - + + snap_input - -snap_input + +snap_input eth_input->snap_input - - + + arp_input - -arp_input + +arp_input eth_input->arp_input - - + + ip_input - -ip_input + +ip_input eth_input->ip_input - - + + ip6_input - -ip6_input + +ip6_input eth_input->ip6_input - - + + + + + +mpls_input + +mpls_input + + + +eth_input->mpls_input + + eth_output - -eth_output + +eth_output - + eth_output->iface_output - - + + l2_redirect - -l2_redirect + +l2_redirect lacp_output - -lacp_output + +lacp_output - + lacp_output->eth_output - - + + - + snap_input->l2_redirect - - + + arp_input_reply - -arp_input_reply + +arp_input_reply - + arp_input->arp_input_reply - - + + arp_input_request - -arp_input_request + +arp_input_request - + arp_input->arp_input_request - - + + arp_output_reply - -arp_output_reply + +arp_output_reply - + arp_output_reply->eth_output - - + + arp_output_request - -arp_output_request + +arp_output_request - + arp_output_request->eth_output - - + + bridge_flood - -bridge_flood + +bridge_flood - + bridge_flood->iface_input - - + + - + bridge_flood->iface_output - - + + vxlan_flood - -vxlan_flood + +vxlan_flood - + bridge_flood->vxlan_flood - - + + - + bridge_input->iface_input - - + + - + bridge_input->iface_output - - + + - + bridge_input->bridge_flood - - + + bridge_neigh_suppress - -bridge_neigh_suppress + +bridge_neigh_suppress - + bridge_input->bridge_neigh_suppress - - + + - + bridge_neigh_suppress->iface_output - - + + - + bridge_neigh_suppress->bridge_flood - - + + - + vxlan_flood->iface_output - - + + ospf_redirect - -ospf_redirect + +ospf_redirect - + ospf_redirect->l2_redirect - - + + loopback_input - -loopback_input + +loopback_input - + loopback_input->ip_input - - + + - + loopback_input->ip6_input - - + + loopback_output - -loopback_output + +loopback_output xvrf - -xvrf + +xvrf - + xvrf->ip_input - - + + - + xvrf->ip6_input - - + + ip_forward - -ip_forward + +ip_forward ip_output - -ip_output + +ip_output - + ip_forward->ip_output - - + + ip_fragment - -ip_fragment + +ip_fragment - + ip_fragment->ip_output - - + + ip_hold - -ip_hold + +ip_hold - + ip_input->ip_forward - - + + ip_input_local - -ip_input_local + +ip_input_local - + ip_input->ip_input_local - - + + - + ip_input->ip_output - - + + - + dnat44_dynamic - -dnat44_dynamic + +dnat44_dynamic - + ip_input->dnat44_dynamic - - + + - + dnat44_static - -dnat44_static + +dnat44_static - + ip_input->dnat44_static - - + + - + ip_input_local->ospf_redirect - - + + ipip_input - -ipip_input + +ipip_input - + ip_input_local->ipip_input - - + + - + icmp_input - -icmp_input + +icmp_input - + ip_input_local->icmp_input - - + + - + l4_input_local - -l4_input_local + +l4_input_local - + ip_input_local->l4_input_local - - + + - + ip_output->eth_output - - + + - + ip_output->xvrf - - + + - + ip_output->ip_fragment - - + + - + ip_output->ip_hold - - + + ipip_output - -ipip_output + +ipip_output - + ip_output->ipip_output - - + + + + + +mpls_push + +mpls_push + + + +ip_output->mpls_push + + - + sr6_output - -sr6_output + +sr6_output - + ip_output->sr6_output - - + + ip6_forward - -ip6_forward + +ip6_forward ip6_output - -ip6_output + +ip6_output - + ip6_forward->ip6_output - - + + ip6_hold - -ip6_hold + +ip6_hold - + ip6_input->ip6_forward - - + + ip6_input_local - -ip6_input_local + +ip6_input_local - + ip6_input->ip6_input_local - - + + - + ip6_input->ip6_output - - + + - + sr6_local - -sr6_local + +sr6_local - + ip6_input->sr6_local - - + + - + ip6_input_local->ospf_redirect - - + + - + icmp6_input - -icmp6_input + +icmp6_input - + ip6_input_local->icmp6_input - - + + - + ip6_input_local->l4_input_local - - + + - + ip6_output->eth_output - - + + - + ip6_output->xvrf - - + + - + ip6_output->ip6_hold - - + + + + + +ip6_output->mpls_push + + - + ip6_output->sr6_output - - + + - + ipip_input->ip_input - - + + - + ipip_output->ip_output - - + + - + +mpls_push_frag_needed + +mpls_push_frag_needed + + + +icmp_output + +icmp_output + + + +mpls_push_frag_needed->icmp_output + + + + + +mpls_push_pkt_too_big + +mpls_push_pkt_too_big + + + +icmp6_output + +icmp6_output + + + +mpls_push_pkt_too_big->icmp6_output + + + + + +mpls_output_frag_needed + +mpls_output_frag_needed + + + +mpls_output_frag_needed->icmp_output + + + + + +mpls_output_pkt_too_big + +mpls_output_pkt_too_big + + + +mpls_output_pkt_too_big->icmp6_output + + + + + +mpls_input->ip_input + + + + + +mpls_input->ip6_input + + + + + +mpls_output + +mpls_output + + + +mpls_input->mpls_output + + + + + +mpls_output->eth_output + + + + + +mpls_output->ip_hold + + + + + +mpls_output->mpls_output_frag_needed + + + + + +mpls_output->mpls_output_pkt_too_big + + + + + +mpls_push->mpls_push_frag_needed + + + + + +mpls_push->mpls_push_pkt_too_big + + + + + +mpls_push->mpls_output + + + + + vxlan_input - -vxlan_input + +vxlan_input - + vxlan_input->iface_input - - + + - + vxlan_output->ip_output - - + + - + vxlan_output->ip6_output - - + + - + dnat44_dynamic->ip_forward - - + + - + dnat44_dynamic->ip_input_local - - + + - + dnat44_static->ip_forward - - + + - + dnat44_static->ip_input_local - - + + - + sr6_local->ip_input - - + + - + sr6_local->ip6_forward - - + + - + sr6_local->ip6_input - - + + - + sr6_local->ip6_input_local - - + + - + sr6_output->ip6_output - - - - - -icmp_output - -icmp_output + + - + icmp_input->icmp_output - - + + - + icmp_local_send - -icmp_local_send + +icmp_local_send - + icmp_local_send->icmp_output - - + + - + icmp_output->ip_output - - - - - -icmp6_output - -icmp6_output + + - + icmp6_input->icmp6_output - - + + - + ndp_na_input - -ndp_na_input + +ndp_na_input - + icmp6_input->ndp_na_input - - + + - + ndp_ns_input - -ndp_ns_input + +ndp_ns_input - + icmp6_input->ndp_ns_input - - + + - + ndp_ra_input - -ndp_ra_input + +ndp_ra_input - + icmp6_input->ndp_ra_input - - + + - + ndp_rs_input - -ndp_rs_input + +ndp_rs_input - + icmp6_input->ndp_rs_input - - + + - + icmp6_local_send - -icmp6_local_send + +icmp6_local_send - + icmp6_local_send->icmp6_output - - + + - + icmp6_output->ip6_output - - + + - + ndp_na_output - -ndp_na_output + +ndp_na_output - + ndp_na_output->icmp6_output - - + + - + ndp_ns_output - -ndp_ns_output + +ndp_ns_output - + ndp_ns_output->icmp6_output - - + + - + l4_loopback_output - -l4_loopback_output + +l4_loopback_output - + ndp_ra_input->l4_loopback_output - - + + - + l4_input_local->vxlan_input - - + + - + l4_input_local->l4_loopback_output - - + + - + dhcp_input - -dhcp_input + +dhcp_input - + l4_input_local->dhcp_input - - + + - + l4_loopback_output->loopback_output - - + + diff --git a/docs/grout-frr.7.scdoc b/docs/grout-frr.7.scdoc index 1386a7025..bdcab0186 100644 --- a/docs/grout-frr.7.scdoc +++ b/docs/grout-frr.7.scdoc @@ -27,7 +27,9 @@ routes to grout, and grout notifies FRR of link and address changes. The following zebra dataplane operations are handled: - IPv4/IPv6 route install, update and delete -- Nexthop install, update and delete +- Nexthop install, update and delete (including labeled nexthops for + MPLS imposition, e.g. BGP-LU, L3VPN and SR-MPLS) +- MPLS LSP install, update and delete (e.g. from LDP) - IP address install and uninstall - MAC/FDB install and delete - VXLAN flood VTEP add and delete diff --git a/frr/meson.build b/frr/meson.build index b8311aede..3714d6d98 100644 --- a/frr/meson.build +++ b/frr/meson.build @@ -61,6 +61,25 @@ if install_build_flag ) endif +if cmocka_dep.found() + # FRR is built as an external subproject and only provides include dirs + # (the dplane_grout plugin inherits FRR symbols from the parent process at + # runtime). For a standalone test binary we must link libfrr explicitly. + frr_prefix = frr_dep.get_variable('prefix') + frr_libdir = frr_prefix / 'lib' + mpls_frr_test_exe = executable( + 'mpls_frr_test', + files('mpls_frr_test.c', 'rt_grout.c') + grout_header, + c_args: frr_c_args + ['-D__GROUT_UNIT_TEST__'], + include_directories: api_inc + include_directories('.'), + dependencies: [frr_dep, cmocka_dep], + link_args: ['-L' + frr_libdir, '-lfrr', '-Wl,-rpath,' + frr_libdir], + install: false, + override_options: ['b_sanitize=none'], + ) + test('mpls_frr_test', mpls_frr_test_exe, suite: 'unit') +endif + systemd_dep = dependency('systemd', required: false) if systemd_dep.found() systemd_system_unit_dir = systemd_dep.get_variable( diff --git a/frr/mpls_frr_test.c b/frr/mpls_frr_test.c new file mode 100644 index 000000000..930656aa2 --- /dev/null +++ b/frr/mpls_frr_test.c @@ -0,0 +1,235 @@ +// SPDX-License-Identifier: GPL-2.0-or-later +// Copyright (c) 2025 Matej Muzila + +#include "rt_grout.h" + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +// Forward declarations matching if_map.h so -Wmissing-prototypes is satisfied. +// These stubs replace if_map.c which is not linked into the test binary. +uint16_t ifindex_frr_to_grout(ifindex_t); +uint16_t vrf_frr_to_grout(vrf_id_t); + +uint16_t ifindex_frr_to_grout(ifindex_t) { + return 0; +} +uint16_t vrf_frr_to_grout(vrf_id_t) { + return 0; +} + +static void test_lsptype2origin(void **) { + assert_int_equal(lsptype2origin(ZEBRA_LSP_STATIC), GR_NH_ORIGIN_ZSTATIC); + assert_int_equal(lsptype2origin(ZEBRA_LSP_LDP), GR_NH_ORIGIN_LDP); + assert_int_equal(lsptype2origin(ZEBRA_LSP_BGP), GR_NH_ORIGIN_BGP); + assert_int_equal(lsptype2origin(ZEBRA_LSP_OSPF_SR), GR_NH_ORIGIN_OSPF); + assert_int_equal(lsptype2origin(ZEBRA_LSP_ISIS_SR), GR_NH_ORIGIN_ISIS); + assert_int_equal(lsptype2origin(ZEBRA_LSP_SHARP), GR_NH_ORIGIN_SHARP); + assert_int_equal(lsptype2origin(ZEBRA_LSP_SRTE), GR_NH_ORIGIN_SRTE); + assert_int_equal(lsptype2origin(ZEBRA_LSP_NONE), GR_NH_ORIGIN_ZEBRA); + assert_int_equal(lsptype2origin(ZEBRA_LSP_EVPN), GR_NH_ORIGIN_ZEBRA); +} + +static void test_nh_has_mpls_labels_null_label(void **) { + struct nexthop nh = {}; + + assert_false(nh_has_mpls_labels(&nh)); +} + +static void test_nh_has_mpls_labels_zero_count(void **) { + struct mpls_label_stack nhl = {.num_labels = 0}; + struct nexthop nh = {}; + + nh.nh_label = &nhl; + assert_false(nh_has_mpls_labels(&nh)); +} + +static void test_nh_has_mpls_labels_implicit_null(void **) { + struct mpls_label_stack *nhl; + struct nexthop nh = {}; + + nhl = malloc(sizeof(*nhl) + sizeof(mpls_label_t)); + + assert_non_null(nhl); + nhl->num_labels = 1; + nhl->label[0] = MPLS_LABEL_IMPLICIT_NULL; + nh.nh_label = nhl; + assert_false(nh_has_mpls_labels(&nh)); + free(nhl); +} + +static void test_nh_has_mpls_labels_real_label(void **) { + struct mpls_label_stack *nhl; + struct nexthop nh = {}; + + nhl = malloc(sizeof(*nhl) + sizeof(mpls_label_t)); + + assert_non_null(nhl); + nhl->num_labels = 1; + nhl->label[0] = 100; + nh.nh_label = nhl; + assert_true(nh_has_mpls_labels(&nh)); + free(nhl); +} + +static void test_nh_has_mpls_labels_two_labels(void **) { + struct mpls_label_stack *nhl; + struct nexthop nh = {}; + + nhl = malloc(sizeof(*nhl) + 2 * sizeof(mpls_label_t)); + + assert_non_null(nhl); + nhl->num_labels = 2; + nhl->label[0] = 100; + nhl->label[1] = 200; + nh.nh_label = nhl; + assert_true(nh_has_mpls_labels(&nh)); + free(nhl); +} + +static void test_fill_mpls_nh_ipv4_one_label(void **) { + struct gr_nexthop_info_mpls *mpls; + struct mpls_label_stack *nhl; + struct gr_nh_add_req *req; + struct nexthop nh = {}; + size_t len; + + len = sizeof(struct gr_nh_add_req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + mpls = (struct gr_nexthop_info_mpls *)req->nh.info; + nhl = malloc(sizeof(*nhl) + sizeof(mpls_label_t)); + + assert_non_null(req); + + nh.type = NEXTHOP_TYPE_IPV4_IFINDEX; + nh.gate.ipv4.s_addr = htonl(0xAC100102); + + assert_non_null(nhl); + nhl->num_labels = 1; + nhl->label[0] = 100; + nh.nh_label = nhl; + + assert_int_equal(grout_fill_mpls_nh(req, 42, GR_NH_ORIGIN_ZSTATIC, &nh), 0); + + assert_int_equal(mpls->via.af, GR_AF_IP4); + assert_int_equal(mpls->n_labels, 1); + assert_int_equal(mpls->labels[0], 100); + assert_int_equal(mpls->ttl, 0); + assert_int_equal(mpls->payload_af, GR_AF_UNSPEC); + assert_true(req->exist_ok); + assert_int_equal(req->nh.nh_id, 42); + assert_int_equal(req->nh.origin, GR_NH_ORIGIN_ZSTATIC); + assert_int_equal(req->nh.type, GR_NH_T_MPLS); + + free(nhl); + free(req); +} + +static void test_fill_mpls_nh_ipv6_two_labels(void **) { + struct gr_nexthop_info_mpls *mpls; + struct mpls_label_stack *nhl; + struct gr_nh_add_req *req; + struct nexthop nh = {}; + size_t len; + + len = sizeof(struct gr_nh_add_req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + mpls = (struct gr_nexthop_info_mpls *)req->nh.info; + nhl = malloc(sizeof(*nhl) + 2 * sizeof(mpls_label_t)); + + assert_non_null(req); + + nh.type = NEXTHOP_TYPE_IPV6_IFINDEX; + nh.gate.ipv6.s6_addr[0] = 0xfe; + nh.gate.ipv6.s6_addr[1] = 0x80; + nh.gate.ipv6.s6_addr[15] = 0x01; + + assert_non_null(nhl); + nhl->num_labels = 2; + nhl->label[0] = 200; + nhl->label[1] = 100; + nh.nh_label = nhl; + + assert_int_equal(grout_fill_mpls_nh(req, 43, GR_NH_ORIGIN_BGP, &nh), 0); + + assert_int_equal(mpls->via.af, GR_AF_IP6); + assert_int_equal(mpls->n_labels, 2); + assert_int_equal(mpls->labels[0], 200); + assert_int_equal(mpls->labels[1], 100); + + free(nhl); + free(req); +} + +static void test_fill_mpls_nh_too_many_labels(void **) { + struct mpls_label_stack *nhl; + struct gr_nh_add_req *req; + struct nexthop nh = {}; + size_t len; + uint8_t n; + + n = GR_MPLS_MAX_LABELS + 1; + len = sizeof(struct gr_nh_add_req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + nhl = malloc(sizeof(*nhl) + n * sizeof(mpls_label_t)); + + assert_non_null(req); + + nh.type = NEXTHOP_TYPE_IPV4; + nh.gate.ipv4.s_addr = htonl(0xAC100102); + + assert_non_null(nhl); + nhl->num_labels = n; + for (uint8_t i = 0; i < n; i++) + nhl->label[i] = 100 + i; + nh.nh_label = nhl; + + assert_int_equal(grout_fill_mpls_nh(req, 44, GR_NH_ORIGIN_ZSTATIC, &nh), -1); + + free(nhl); + free(req); +} + +static void test_fill_mpls_nh_unsupported_type(void **) { + struct gr_nh_add_req *req; + struct nexthop nh = {}; + size_t len; + + len = sizeof(struct gr_nh_add_req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + + assert_non_null(req); + + nh.type = NEXTHOP_TYPE_BLACKHOLE; + + assert_int_equal(grout_fill_mpls_nh(req, 45, GR_NH_ORIGIN_ZSTATIC, &nh), -1); + + free(req); +} + +int main(void) { + const struct CMUnitTest tests[] = { + cmocka_unit_test(test_lsptype2origin), + cmocka_unit_test(test_nh_has_mpls_labels_null_label), + cmocka_unit_test(test_nh_has_mpls_labels_zero_count), + cmocka_unit_test(test_nh_has_mpls_labels_implicit_null), + cmocka_unit_test(test_nh_has_mpls_labels_real_label), + cmocka_unit_test(test_nh_has_mpls_labels_two_labels), + cmocka_unit_test(test_fill_mpls_nh_ipv4_one_label), + cmocka_unit_test(test_fill_mpls_nh_ipv6_two_labels), + cmocka_unit_test(test_fill_mpls_nh_too_many_labels), + cmocka_unit_test(test_fill_mpls_nh_unsupported_type), + }; + return cmocka_run_group_tests(tests, NULL, NULL); +} diff --git a/frr/rt_grout.c b/frr/rt_grout.c index 8c5bdb32b..b125dfb96 100644 --- a/frr/rt_grout.c +++ b/frr/rt_grout.c @@ -7,9 +7,11 @@ #include "rt_grout.h" #include +#include #include #include +#include #include #include #include @@ -17,8 +19,25 @@ #include #include #include +#include #include +// Synthesize a deterministic grout nexthop ID for an MPLS label route. Label +// routes are 1:1 with their incoming label, so the label doubles as the key. +// The 0x80000000 offset keeps these IDs clear of both grout's allocation pool +// (<= 1 << 17) and FRR's proto-NHG range (< ZEBRA_NHG_PROTO_UPPER, ~250M). +#define GROUT_MPLS_NH_ID(label) (0x80000000u | (uint32_t)(label)) + +// Under __GROUT_UNIT_TEST__, expose the three MPLS helper functions so that +// mpls_frr_test.c can call them directly without the full plugin link context. +#ifdef __GROUT_UNIT_TEST__ +#define TESTABLE_STATIC +#else +#define TESTABLE_STATIC static +#endif + +#ifndef __GROUT_UNIT_TEST__ + static inline bool is_selfroute(gr_nh_origin_t origin) { switch (origin) { case GR_NH_ORIGIN_ZEBRA: @@ -507,6 +526,20 @@ void grout_route6_change(bool new, struct gr_ip6_route *gr_r6, bool startup) { ); } +void grout_mpls_route_change(bool new, const struct gr_mpls_label_route *route, bool /*startup*/) { + gr_log_debug( + "%s in_label %u vrf %u nh_id %u", + new ? "add" : "del", + route->in_label, + route->vrf_id, + route->nh_id + ); + // MPLS route events from grout (e.g. statically added via grcli) are + // informational only — FRR is the label distribution protocol and owns + // the LFIB. A full ZAPI push would require lsp_add/del_nhlfe which is + // internal to zebra_mpls. +} + enum zebra_dplane_result grout_add_del_route(struct zebra_dplane_ctx *ctx) { union { struct gr_ip4_route_add_req r4_add; @@ -627,6 +660,181 @@ enum zebra_dplane_result grout_add_del_route(struct zebra_dplane_ctx *ctx) { return ZEBRA_DPLANE_REQUEST_SUCCESS; } +#endif /* !__GROUT_UNIT_TEST__ */ + +TESTABLE_STATIC inline gr_nh_origin_t lsptype2origin(enum lsp_types_t type) { + switch (type) { + case ZEBRA_LSP_STATIC: + return GR_NH_ORIGIN_ZSTATIC; + case ZEBRA_LSP_LDP: + return GR_NH_ORIGIN_LDP; + case ZEBRA_LSP_BGP: + return GR_NH_ORIGIN_BGP; + case ZEBRA_LSP_OSPF_SR: + return GR_NH_ORIGIN_OSPF; + case ZEBRA_LSP_ISIS_SR: + return GR_NH_ORIGIN_ISIS; + case ZEBRA_LSP_SHARP: + return GR_NH_ORIGIN_SHARP; + case ZEBRA_LSP_SRTE: + return GR_NH_ORIGIN_SRTE; + case ZEBRA_LSP_NONE: + case ZEBRA_LSP_EVPN: + default: + return GR_NH_ORIGIN_ZEBRA; + } +} + +// A lone implicit-null is penultimate hop popping, which grout +// represents as an MPLS nexthop with zero output labels, so it does not count +// as "having labels" here. +TESTABLE_STATIC bool nh_has_mpls_labels(const struct nexthop *nh) { + const struct mpls_label_stack *nhl = nh->nh_label; + + return nhl != NULL && nhl->num_labels > 0 + && !(nhl->num_labels == 1 && nhl->label[0] == MPLS_LABEL_IMPLICIT_NULL); +} + +// Populate an MPLS nexthop add request from a FRR nexthop carrying output +// labels. Returns 0 on success, -1 if the nexthop cannot be represented in +// grout. The caller must allocate req with room for a gr_nexthop_info_mpls. +TESTABLE_STATIC int grout_fill_mpls_nh( + struct gr_nh_add_req *req, + uint32_t nh_id, + gr_nh_origin_t origin, + const struct nexthop *nh +) { + struct gr_nexthop_info_mpls *mpls = (struct gr_nexthop_info_mpls *)req->nh.info; + + req->exist_ok = true; + req->nh.nh_id = nh_id; + req->nh.origin = origin; + req->nh.type = GR_NH_T_MPLS; + req->nh.vrf_id = vrf_frr_to_grout(nh->vrf_id); + req->nh.iface_id = ifindex_frr_to_grout(nh->ifindex); + + // grout resolves the outgoing L2 header through an L3 gateway, so an + // MPLS nexthop must carry one. Extract it from the FRR nexthop. + switch (nh->type) { + case NEXTHOP_TYPE_IPV4: + case NEXTHOP_TYPE_IPV4_IFINDEX: + mpls->via.af = GR_AF_IP4; + memcpy(&mpls->via.ipv4, &nh->gate.ipv4, sizeof(mpls->via.ipv4)); + break; + case NEXTHOP_TYPE_IPV6: + case NEXTHOP_TYPE_IPV6_IFINDEX: + mpls->via.af = GR_AF_IP6; + memcpy(&mpls->via.ipv6, &nh->gate.ipv6, sizeof(mpls->via.ipv6)); + break; + default: + gr_log_err("MPLS nexthop requires an IP gateway, type %u unsupported", nh->type); + return -1; + } + + // Copy the output label stack. A lone implicit-null (penultimate hop + // popping) leaves n_labels at 0, i.e. pop with no imposition. Explicit + // null (0/2) is a real label and stays on the wire. + if (nh_has_mpls_labels(nh)) { + const struct mpls_label_stack *nhl = nh->nh_label; + + if (nhl->num_labels > GR_MPLS_MAX_LABELS) { + gr_log_err("too many output labels: %u", nhl->num_labels); + return -1; + } + mpls->n_labels = nhl->num_labels; + for (uint8_t i = 0; i < nhl->num_labels; i++) + mpls->labels[i] = nhl->label[i]; + } + + mpls->ttl = 0; // copy TTL from the incoming label / payload + mpls->payload_af = GR_AF_UNSPEC; // auto-detect payload after pop + + return 0; +} + +#ifndef __GROUT_UNIT_TEST__ + +enum zebra_dplane_result grout_add_del_lsp(struct zebra_dplane_ctx *ctx) { + struct gr_mpls_label_route_del_req del; + struct gr_mpls_label_route_add_req add; + const struct zebra_nhlfe *best; + struct gr_nh_del_req nh_del; + struct gr_nh_add_req *req; + mpls_label_t in_label; + gr_nh_origin_t origin; + uint32_t vrf_id; + uint32_t nh_id; + size_t len; + bool new; + + vrf_id = vrf_frr_to_grout(dplane_ctx_get_vrf(ctx)); + new = dplane_ctx_get_op(ctx) != DPLANE_OP_LSP_DELETE; + in_label = dplane_ctx_get_in_label(ctx); + nh_id = GROUT_MPLS_NH_ID(in_label); + + gr_log_debug("%s in_label %u vrf %u", new ? "add" : "del", in_label, vrf_id); + + if (in_label == MPLS_INVALID_LABEL || in_label > GR_MPLS_LABEL_MAX) { + gr_log_err("invalid in_label %u, skip", in_label); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + + if (!new) { + del = (struct gr_mpls_label_route_del_req) { + .vrf_id = vrf_id, + .in_label = in_label, + .missing_ok = true, + }; + nh_del = (struct gr_nh_del_req) {.missing_ok = true, .nh = {.nh_id = nh_id}}; + + // Drop the label route first so nothing references the nexthop, + // then remove the synthesized nexthop. + if (grout_client_send_recv(GR_MPLS_LABEL_ROUTE_DEL, sizeof(del), &del, NULL) < 0) + return ZEBRA_DPLANE_REQUEST_FAILURE; + if (grout_client_send_recv(GR_NH_DEL, sizeof(nh_del), &nh_del, NULL) < 0) + return ZEBRA_DPLANE_REQUEST_FAILURE; + return ZEBRA_DPLANE_REQUEST_SUCCESS; + } + + best = dplane_ctx_get_best_nhlfe(ctx); + if (best == NULL || best->nexthop == NULL) { + gr_log_err("LSP %u has no best nexthop, skip", in_label); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + + origin = lsptype2origin(best->type); + + len = sizeof(*req) + sizeof(struct gr_nexthop_info_mpls); + req = calloc(1, len); + if (req == NULL) { + gr_log_err("calloc: %s", strerror(errno)); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + + if (grout_fill_mpls_nh(req, nh_id, origin, best->nexthop) < 0) { + free(req); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + + if (grout_client_send_recv(GR_NH_ADD, len, req, NULL) < 0) { + free(req); + return ZEBRA_DPLANE_REQUEST_FAILURE; + } + free(req); + + add = (struct gr_mpls_label_route_add_req) { + .vrf_id = vrf_id, + .in_label = in_label, + .nh_id = nh_id, + .origin = origin, + .exist_ok = true, + }; + if (grout_client_send_recv(GR_MPLS_LABEL_ROUTE_ADD, sizeof(add), &add, NULL) < 0) + return ZEBRA_DPLANE_REQUEST_FAILURE; + + return ZEBRA_DPLANE_REQUEST_SUCCESS; +} + static enum zebra_dplane_result grout_add_nexthop_group(struct zebra_dplane_ctx *ctx) { enum zebra_dplane_result ret = ZEBRA_DPLANE_REQUEST_SUCCESS; uint32_t nh_id = dplane_ctx_get_nhe_id(ctx); @@ -702,6 +910,9 @@ grout_add_nexthop(uint32_t nh_id, gr_nh_origin_t origin, const struct nexthop *n len += sizeof(*sr6) + nh->nh_srv6->seg6_segs->num_segs * sizeof(sr6->seglist[0]); type = GR_NH_T_SR6_OUTPUT; + } else if (nh_has_mpls_labels(nh)) { + len += sizeof(struct gr_nexthop_info_mpls); + type = GR_NH_T_MPLS; } else { len += sizeof(*l3); type = GR_NH_T_L3; @@ -731,6 +942,10 @@ grout_add_nexthop(uint32_t nh_id, gr_nh_origin_t origin, const struct nexthop *n req->nh.iface_id = ifindex_frr_to_grout(nh->ifindex); switch (type) { + case GR_NH_T_MPLS: + if (grout_fill_mpls_nh(req, nh_id, origin, nh) < 0) + goto out; + break; case GR_NH_T_L3: // For L3 nexthops in VRFs with an L3VNI, redirect the iface from // the VRF (SVI in FRR's model) to the VXLAN interface. Grout @@ -1448,3 +1663,4 @@ enum zebra_dplane_result grout_neigh_read_ctx(struct zebra_dplane_ctx *ctx) { return ZEBRA_DPLANE_REQUEST_SUCCESS; } #endif +#endif /* !__GROUT_UNIT_TEST__ */ diff --git a/frr/rt_grout.h b/frr/rt_grout.h index a6f08bfdd..f471806f1 100644 --- a/frr/rt_grout.h +++ b/frr/rt_grout.h @@ -6,12 +6,15 @@ #include #include #include +#include #include void grout_route4_change(bool new, struct gr_ip4_route *gr_r4, bool startup); void grout_route6_change(bool new, struct gr_ip6_route *gr_r6, bool startup); +void grout_mpls_route_change(bool new, const struct gr_mpls_label_route *route, bool /*startup*/); enum zebra_dplane_result grout_add_del_route(struct zebra_dplane_ctx *ctx); +enum zebra_dplane_result grout_add_del_lsp(struct zebra_dplane_ctx *ctx); enum zebra_dplane_result grout_add_del_nexthop(struct zebra_dplane_ctx *ctx); void grout_nexthop_change(bool new, struct gr_nexthop *gr_nh, bool startup); void grout_nexthop_group_add(struct gr_nexthop *gr_nh, bool startup); @@ -23,3 +26,16 @@ enum zebra_dplane_result grout_neigh_update_ctx(struct zebra_dplane_ctx *ctx); enum zebra_dplane_result grout_vxlan_flood_update_ctx(struct zebra_dplane_ctx *ctx); enum zebra_dplane_result grout_fdb_read_ctx(struct zebra_dplane_ctx *ctx); enum zebra_dplane_result grout_neigh_read_ctx(struct zebra_dplane_ctx *ctx); + +#ifdef __GROUT_UNIT_TEST__ +#include +#include +gr_nh_origin_t lsptype2origin(enum lsp_types_t type); +bool nh_has_mpls_labels(const struct nexthop *nh); +int grout_fill_mpls_nh( + struct gr_nh_add_req *req, + uint32_t nh_id, + gr_nh_origin_t origin, + const struct nexthop *nh +); +#endif diff --git a/frr/zebra_dplane_grout.c b/frr/zebra_dplane_grout.c index 2f901f678..e788cdc92 100644 --- a/frr/zebra_dplane_grout.c +++ b/frr/zebra_dplane_grout.c @@ -13,6 +13,8 @@ #include "log_grout.h" #include "rt_grout.h" +#include + #include #include #include @@ -90,6 +92,8 @@ static void zebra_grout_connect(struct event *); static void grout_sync(struct event *); static void grout_sync_ifaces(struct event *); static void grout_sync_addrs(struct event *); +static void grout_sync_lsps(struct event *); +static void grout_sync_routes(struct event *); static void grout_reconnect(struct event *); static void grout_reconnect_finish(void); static void grout_main_router_started(void); @@ -380,6 +384,53 @@ static void grout_sync_inject_marker(void) { ); } +static void grout_sync_lsps(struct event *e) { + struct gr_mpls_label_route_list_req req; + struct gr_mpls_label_route *route; + int ret; + + req = (struct gr_mpls_label_route_list_req) {.vrf_id = EVENT_VAL(e), .max_count = 0}; + + gr_log_info("vrf %u", EVENT_VAL(e)); + + gr_api_client_stream_foreach ( + route, ret, grout_ctx.sync_client, GR_MPLS_LABEL_ROUTE_LIST, sizeof(req), &req + ) { + grout_mpls_route_change(true, route, true); + } + if (ret < 0) { + gr_log_err("GR_MPLS_LABEL_ROUTE_LIST: %s", strerror(errno)); + event_add_timer( + zrouter.master, grout_reconnect, NULL, 1, &grout_ctx.dg_t_zebra_sync + ); + return; + } + + // Chain to next VRF's LSPs. + for (unsigned int i = EVENT_VAL(e) + 1; i < grout_ctx.max_ifaces; i++) { + if (bf_test_index(grout_ctx.sync_vrf, i)) { + event_add_event( + zrouter.master, grout_sync_lsps, NULL, i, &grout_ctx.dg_t_zebra_sync + ); + return; + } + } + + // All VRFs' LSPs done. Kick off Pass 4 (routes) from the first VRF. + for (unsigned int i = 0; i < grout_ctx.max_ifaces; i++) { + if (bf_test_index(grout_ctx.sync_vrf, i)) { + event_add_event( + zrouter.master, + grout_sync_routes, + NULL, + i, + &grout_ctx.dg_t_zebra_sync + ); + return; + } + } +} + static void grout_sync_routes(struct event *e) { struct gr_ip4_route_list_req r4_req = {.vrf_id = EVENT_VAL(e), .max_count = 0}; struct gr_ip4_route *r4; @@ -479,15 +530,11 @@ static void grout_sync_nh_groups(struct event *) { return; } - // Kick off routes starting from the first VRF. + // Kick off LSPs starting from the first VRF. for (unsigned int i = 0; i < grout_ctx.max_ifaces; i++) { if (bf_test_index(grout_ctx.sync_vrf, i)) { event_add_event( - zrouter.master, - grout_sync_routes, - NULL, - i, - &grout_ctx.dg_t_zebra_sync + zrouter.master, grout_sync_lsps, NULL, i, &grout_ctx.dg_t_zebra_sync ); return; } @@ -530,7 +577,7 @@ static void grout_sync_nhs(struct event *e) { } // Individual NHs done across all VRFs. Sync NH groups (global, not - // per-VRF) so that group member references resolve before routes. + // per-VRF) so that group member references resolve before LSPs and routes. event_add_event(zrouter.master, grout_sync_nh_groups, NULL, 0, &grout_ctx.dg_t_zebra_sync); } @@ -733,6 +780,8 @@ static void zebra_grout_connect(struct event *) { {.type = GR_EVENT_NEXTHOP_NEW, .suppress_self_events = true}, {.type = GR_EVENT_NEXTHOP_DELETE, .suppress_self_events = true}, {.type = GR_EVENT_NEXTHOP_UPDATE, .suppress_self_events = true}, + {.type = GR_EVENT_MPLS_ROUTE_ADD, .suppress_self_events = true}, + {.type = GR_EVENT_MPLS_ROUTE_DEL, .suppress_self_events = true}, }; if (grout_notif_subscribe(&grout_ctx.zebra_notifs, gr_evts, ARRAY_DIM(gr_evts)) < 0) { @@ -930,6 +979,12 @@ static void zebra_read_notifications(struct event *event) { case GR_EVENT_NEXTHOP_DELETE: grout_nexthop_change(new, PAYLOAD(gr_e), false); break; + case GR_EVENT_MPLS_ROUTE_ADD: + new = true; + // fallthrough + case GR_EVENT_MPLS_ROUTE_DEL: + grout_mpls_route_change(new, PAYLOAD(gr_e), false); + break; } free(gr_e); @@ -956,6 +1011,11 @@ static enum zebra_dplane_result zd_grout_process_update(struct zebra_dplane_ctx case DPLANE_OP_NH_DELETE: return grout_add_del_nexthop(ctx); + case DPLANE_OP_LSP_INSTALL: + case DPLANE_OP_LSP_UPDATE: + case DPLANE_OP_LSP_DELETE: + return grout_add_del_lsp(ctx); + case DPLANE_OP_MAC_INSTALL: case DPLANE_OP_MAC_DELETE: return grout_macfdb_update_ctx(ctx); diff --git a/meson.build b/meson.build index 842c2a032..8296da47b 100644 --- a/meson.build +++ b/meson.build @@ -162,6 +162,9 @@ subdir('main') subdir('modules') subdir('cli') subdir('api') + +cmocka_dep = dependency('cmocka', required: get_option('tests')) + subdir('frr') fs = import('fs') @@ -224,7 +227,6 @@ pkg.generate( install_dir: get_option('datadir') / 'pkgconfig', ) -cmocka_dep = dependency('cmocka', required: get_option('tests')) if cmocka_dep.found() foreach t : tests name = fs.replace_suffix(t['sources'].get(0), '').underscorify() diff --git a/modules/infra/api/gr_nexthop.h b/modules/infra/api/gr_nexthop.h index ef9497914..e3f7ffcab 100644 --- a/modules/infra/api/gr_nexthop.h +++ b/modules/infra/api/gr_nexthop.h @@ -38,6 +38,7 @@ typedef enum : uint8_t { GR_NH_T_BLACKHOLE, // Drop packets silently. GR_NH_T_REJECT, // Drop packets with ICMP error. GR_NH_T_GROUP, // ECMP for multipath routing. + GR_NH_T_MPLS, // MPLS label imposition/swap. #define GR_NH_T_ALL UINT8_C(0xff) // Match all types in list operations. } gr_nh_type_t; @@ -199,6 +200,8 @@ static inline const char *gr_nh_type_name(const gr_nh_type_t type) { return "reject"; case GR_NH_T_GROUP: return "group"; + case GR_NH_T_MPLS: + return "MPLS"; } return "?"; } diff --git a/modules/infra/control/l3_nexthop.c b/modules/infra/control/l3_nexthop.c index 6944839b7..c4892ad31 100644 --- a/modules/infra/control/l3_nexthop.c +++ b/modules/infra/control/l3_nexthop.c @@ -45,6 +45,8 @@ const struct nexthop_af_ops *nexthop_af_ops_from_mbuf(const struct rte_mbuf *m) return af_ops[GR_AF_IP4]; if (m->packet_type & RTE_PTYPE_L3_IPV6) return af_ops[GR_AF_IP6]; + if ((m->packet_type & RTE_PTYPE_TUNNEL_MASK) == RTE_PTYPE_TUNNEL_MPLS_IN_GRE) + return af_ops[GR_AF_MPLS]; return NULL; } @@ -78,6 +80,9 @@ static inline void set_nexthop_key( key->ipv6.a[3] = iface_id & 0xff; } break; + case GR_AF_MPLS: + ABORT("AF_MPLS has no nexthop key with gw"); + break; case GR_AF_UNSPEC: ABORT("AF_UNSPEC has no nexthop key with gw"); break; @@ -206,6 +211,7 @@ static bool l3_equal(const struct nexthop *a, const struct nexthop *b) { case GR_AF_IP6: return rte_ipv6_addr_eq(&l3_a->ipv6, &l3_b->ipv6); case GR_AF_UNSPEC: + case GR_AF_MPLS: return true; } return false; diff --git a/modules/infra/control/nexthop.c b/modules/infra/control/nexthop.c index d4549f39e..6a33a5010 100644 --- a/modules/infra/control/nexthop.c +++ b/modules/infra/control/nexthop.c @@ -240,6 +240,7 @@ bool nexthop_type_valid(gr_nh_type_t type) { case GR_NH_T_BLACKHOLE: case GR_NH_T_REJECT: case GR_NH_T_GROUP: + case GR_NH_T_MPLS: return true; } return false; diff --git a/modules/infra/control/vrf.h b/modules/infra/control/vrf.h index 428ea6e34..734d2214a 100644 --- a/modules/infra/control/vrf.h +++ b/modules/infra/control/vrf.h @@ -13,6 +13,7 @@ GR_IFACE_INFO(GR_IFACE_TYPE_VRF, iface_info_vrf, { uint32_t vrf_ifindex; void *fib4; void *fib6; + void *fib_mpls; }); // Map a grout VRF ID to the kernel routing table ID. diff --git a/modules/infra/datapath/checksum.h b/modules/infra/datapath/checksum.h new file mode 100644 index 000000000..71387f560 --- /dev/null +++ b/modules/infra/datapath/checksum.h @@ -0,0 +1,33 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#pragma once + +#include + +// RFC 1624 incremental checksum update for a 16-bit field. +static inline rte_be16_t +fixup_checksum_16(rte_be16_t old_cksum, rte_be16_t old_field, rte_be16_t new_field) { + uint32_t sum; + + sum = ~old_cksum & 0xffff; + sum += (~old_field & 0xffff) + new_field; + sum = (sum >> 16) + (sum & 0xffff); + sum += (sum >> 16); + + return ~sum & 0xffff; +} + +// RFC 1624 incremental checksum update for a 32-bit field. +static inline rte_be16_t +fixup_checksum_32(rte_be16_t old_cksum, ip4_addr_t old_addr, ip4_addr_t new_addr) { + uint32_t sum; + + sum = ~old_cksum & 0xffff; + sum += (~old_addr & 0xffff) + (new_addr & 0xffff); + sum += (~old_addr >> 16) + (new_addr >> 16); + sum = (sum >> 16) + (sum & 0xffff); + sum += (sum >> 16); + + return ~sum & 0xffff; +} diff --git a/modules/ip/datapath/ip_error.c b/modules/ip/datapath/ip_error.c index d4cb27233..b5ff96464 100644 --- a/modules/ip/datapath/ip_error.c +++ b/modules/ip/datapath/ip_error.c @@ -84,7 +84,15 @@ ip_error_process(struct rte_graph *graph, struct rte_node *node, void **objs, ui icmp->icmp_code = ctx->icmp_code; icmp->icmp_cksum = 0; icmp->icmp_ident = 0; - icmp->icmp_seq_nb = 0; + if (ctx->icmp_code == RTE_ICMP_CODE_UNREACH_FRAG) { + // RFC 1191: next-hop MTU in the seq_nb field position + const struct iface *err_iface = mbuf_data(mbuf)->iface; + icmp->icmp_seq_nb = (err_iface != NULL) ? + rte_cpu_to_be_16(err_iface->mtu) : + 0; + } else { + icmp->icmp_seq_nb = 0; + } edge = ICMP_OUTPUT; next: diff --git a/modules/ip6/datapath/ip6_error.c b/modules/ip6/datapath/ip6_error.c index 16fde4772..777293650 100644 --- a/modules/ip6/datapath/ip6_error.c +++ b/modules/ip6/datapath/ip6_error.c @@ -26,10 +26,12 @@ enum edges { static uint16_t ip6_error_process(struct rte_graph *graph, struct rte_node *node, void **objs, uint16_t nb_objs) { const struct ip6_error_ctx *ctx = ip6_error_ctx(node); + struct icmp6_err_pkt_too_big *ptb; struct icmp6_err_dest_unreach *du; struct icmp6_err_ttl_exceeded *te; const struct nexthop_info_l3 *l3; struct ip6_local_mbuf_data *d; + const struct iface *err_iface; const struct iface *iface; const struct nexthop *nh; struct rte_ipv6_hdr *ip; @@ -69,6 +71,15 @@ ip6_error_process(struct rte_graph *graph, struct rte_node *node, void **objs, u goto next; } break; + case ICMP6_ERR_PKT_TOO_BIG: + ptb = gr_mbuf_prepend(mbuf, ptb); + if (unlikely(ptb == NULL)) { + edge = NO_HEADROOM; + goto next; + } + err_iface = mbuf_data(mbuf)->iface; + ptb->mtu = (err_iface != NULL) ? rte_cpu_to_be_32(err_iface->mtu) : 0; + break; default: ABORT("unexpected icmp_type value %hhu", ctx->icmp_type); break; @@ -120,6 +131,15 @@ static int no_route_init(const struct rte_graph *, struct rte_node *node) { return 0; } +static int pkt_too_big_init(const struct rte_graph *, struct rte_node *node) { + struct ip6_error_ctx *ctx; + + ctx = ip6_error_ctx(node); + ctx->icmp_type = ICMP6_ERR_PKT_TOO_BIG; + ctx->icmp_code = 0; + return 0; +} + static struct rte_node_register dest_unreach_node = { .name = "ip6_error_dest_unreach", .process = ip6_error_process, @@ -144,6 +164,18 @@ static struct rte_node_register ttl_exceeded_node = { .init = ttl_exceeded_init, }; +static struct rte_node_register pkt_too_big_node = { + .name = "ip6_error_pkt_too_big", + .process = ip6_error_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp6_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, + .init = pkt_too_big_init, +}; + static struct gr_node_info dest_unreach_info = { .node = &dest_unreach_node, .type = GR_NODE_T_L3, @@ -154,5 +186,11 @@ static struct gr_node_info ttl_exceeded_info = { .type = GR_NODE_T_L3, }; +static struct gr_node_info pkt_too_big_info = { + .node = &pkt_too_big_node, + .type = GR_NODE_T_L3, +}; + GR_NODE_REGISTER(dest_unreach_info); GR_NODE_REGISTER(ttl_exceeded_info); +GR_NODE_REGISTER(pkt_too_big_info); diff --git a/modules/ip6/datapath/ip6_output.c b/modules/ip6/datapath/ip6_output.c index 4853fc3cc..8e0972551 100644 --- a/modules/ip6/datapath/ip6_output.c +++ b/modules/ip6/datapath/ip6_output.c @@ -94,6 +94,8 @@ ip6_output_process(struct rte_graph *graph, struct rte_node *node, void **objs, goto next; } + mbuf_data(mbuf)->iface = iface; + if (rte_pktmbuf_pkt_len(mbuf) > iface->mtu) { edge = TOO_BIG; goto next; @@ -102,7 +104,6 @@ ip6_output_process(struct rte_graph *graph, struct rte_node *node, void **objs, // Determine what is the next node based on the output interface type // By default, it will be eth_output unless another output node was registered. edge = iface_type_edges[iface->type]; - mbuf_data(mbuf)->iface = iface; if (edge != ETH_OUTPUT) goto next; @@ -157,7 +158,7 @@ static struct rte_node_register output_node = { [HOLD] = "ip6_hold", [ERROR] = "ip6_output_error", [DEST_UNREACH] = "ip6_error_dest_unreach", - [TOO_BIG] = "ip6_output_too_big", + [TOO_BIG] = "ip6_error_pkt_too_big", }, }; @@ -171,4 +172,3 @@ static struct gr_node_info info = { GR_NODE_REGISTER(info); GR_DROP_REGISTER(ip6_output_error); -GR_DROP_REGISTER(ip6_output_too_big); diff --git a/modules/l2/control/vxlan.c b/modules/l2/control/vxlan.c index 3d2a7367b..8005b9b31 100644 --- a/modules/l2/control/vxlan.c +++ b/modules/l2/control/vxlan.c @@ -207,6 +207,7 @@ static int iface_vxlan_reconfig( cur->template.ipv6.vxlan.vx_vni = vxlan_encode_vni(cur->vni); break; case GR_AF_UNSPEC: + case GR_AF_MPLS: break; } diff --git a/modules/meson.build b/modules/meson.build index 0b978a62b..8ddb1b3a2 100644 --- a/modules/meson.build +++ b/modules/meson.build @@ -5,6 +5,7 @@ subdir('infra') subdir('ip') subdir('ip6') subdir('ipip') +subdir('mpls') subdir('l2') subdir('l4') subdir('policy') diff --git a/modules/mpls/api/gr_mpls.h b/modules/mpls/api/gr_mpls.h new file mode 100644 index 000000000..020f061b5 --- /dev/null +++ b/modules/mpls/api/gr_mpls.h @@ -0,0 +1,145 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#pragma once + +#include +#include +#include + +#include + +#define GR_MPLS_MODULE 0xf00e + +#define GR_MPLS_MAX_LABELS 16 + +#define GR_MPLS_MAX_STACK_DEPTH 30 + +#define GR_MPLS_LABEL_MAX 0xFFFFF + +// Reserved MPLS label values (RFC 3032). +enum gr_mpls_reserved_labels : uint32_t { + GR_MPLS_LABEL_IPV4_EXPLICIT_NULL = 0, + GR_MPLS_LABEL_ROUTER_ALERT = 1, + GR_MPLS_LABEL_IPV6_EXPLICIT_NULL = 2, + GR_MPLS_LABEL_IMPLICIT_NULL = 3, + GR_MPLS_LABEL_FIRST_UNRESERVED = 16, +}; + +struct gr_nexthop_info_mpls { + uint8_t n_labels; + uint8_t ttl; // 0 = copy from payload. + addr_family_t payload_af; // Payload type after pop (GR_AF_UNSPEC = auto-detect). + struct l3_addr via; + uint32_t labels[GR_MPLS_MAX_LABELS]; +}; + +// label routes +enum gr_mpls_requests : uint32_t { + GR_MPLS_LABEL_ROUTE_ADD = GR_MSG_TYPE(GR_MPLS_MODULE, 0x0001), + GR_MPLS_LABEL_ROUTE_DEL, + GR_MPLS_LABEL_ROUTE_GET, + GR_MPLS_LABEL_ROUTE_LIST, +}; + +// MPLS label route entry. +struct gr_mpls_label_route { + uint16_t vrf_id; + uint32_t in_label; + uint32_t nh_id; + gr_nh_origin_t origin; +}; + +// Add a label route to the LFIB. +struct gr_mpls_label_route_add_req { + uint16_t vrf_id; + uint32_t in_label; // 0 to GR_MPLS_LABEL_MAX. + uint32_t nh_id; // Must reference a GR_NH_T_MPLS nexthop. + gr_nh_origin_t origin; + uint8_t exist_ok; +}; + +GR_REQ(GR_MPLS_LABEL_ROUTE_ADD, struct gr_mpls_label_route_add_req, struct gr_empty); + +// Delete a label route from the LFIB. +struct gr_mpls_label_route_del_req { + uint16_t vrf_id; + uint32_t in_label; + uint8_t missing_ok; +}; + +GR_REQ(GR_MPLS_LABEL_ROUTE_DEL, struct gr_mpls_label_route_del_req, struct gr_empty); + +// Get a single label route by label value. +struct gr_mpls_label_route_get_req { + uint16_t vrf_id; + uint32_t in_label; +}; + +GR_REQ(GR_MPLS_LABEL_ROUTE_GET, struct gr_mpls_label_route_get_req, struct gr_mpls_label_route); + +// List all label routes in a VRF. +struct gr_mpls_label_route_list_req { + uint16_t vrf_id; + uint16_t max_count; +}; + +GR_REQ_STREAM( + GR_MPLS_LABEL_ROUTE_LIST, + struct gr_mpls_label_route_list_req, + struct gr_mpls_label_route +); + +// events + +enum gr_mpls_events : uint32_t { + GR_EVENT_MPLS_ROUTE_ADD = GR_MSG_TYPE(GR_MPLS_MODULE, 0x1001), + GR_EVENT_MPLS_ROUTE_DEL, +}; + +GR_EVENT(GR_EVENT_MPLS_ROUTE_ADD, struct gr_mpls_label_route); +GR_EVENT(GR_EVENT_MPLS_ROUTE_DEL, struct gr_mpls_label_route); + +// label range reservation + +enum gr_mpls_label_range_requests : uint32_t { + GR_MPLS_LABEL_RANGE_ADD = GR_MSG_TYPE(GR_MPLS_MODULE, 0x0010), + GR_MPLS_LABEL_RANGE_DEL, + GR_MPLS_LABEL_RANGE_LIST, +}; + +struct gr_mpls_label_range { + uint16_t vrf_id; + uint32_t start; + uint32_t end; + gr_nh_origin_t origin; +}; + +struct gr_mpls_label_range_add_req { + uint16_t vrf_id; + uint32_t start; + uint32_t end; + gr_nh_origin_t origin; + uint8_t exist_ok; +}; + +GR_REQ(GR_MPLS_LABEL_RANGE_ADD, struct gr_mpls_label_range_add_req, struct gr_empty); + +struct gr_mpls_label_range_del_req { + uint16_t vrf_id; + uint32_t start; + uint32_t end; + uint8_t missing_ok; +}; + +GR_REQ(GR_MPLS_LABEL_RANGE_DEL, struct gr_mpls_label_range_del_req, struct gr_empty); + +struct gr_mpls_label_range_list_req { + uint16_t vrf_id; +}; + +GR_REQ_STREAM( + GR_MPLS_LABEL_RANGE_LIST, + struct gr_mpls_label_range_list_req, + struct gr_mpls_label_range +); diff --git a/modules/mpls/api/meson.build b/modules/mpls/api/meson.build new file mode 100644 index 000000000..c15b0d016 --- /dev/null +++ b/modules/mpls/api/meson.build @@ -0,0 +1,5 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +api_headers += files('gr_mpls.h') +api_inc += include_directories('.') diff --git a/modules/mpls/cli/label.c b/modules/mpls/cli/label.c new file mode 100644 index 000000000..a74b52d6b --- /dev/null +++ b/modules/mpls/cli/label.c @@ -0,0 +1,226 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "cli.h" +#include "cli_event.h" +#include "cli_iface.h" +#include "display.h" + +#include +#include +#include + +#include + +#include + +static cmd_status_t mpls_route_add(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_mpls_label_route_add_req req; + + req = (struct gr_mpls_label_route_add_req) { + .exist_ok = true, + .origin = GR_NH_ORIGIN_STATIC, + }; + + if (arg_u32(p, "LABEL", &req.in_label) < 0) + return CMD_ERROR; + if (arg_u32(p, "ID", &req.nh_id) < 0) + return CMD_ERROR; + if (arg_vrf(c, p, "VRF", &req.vrf_id) < 0) + return CMD_ERROR; + + if (gr_api_client_send_recv(c, GR_MPLS_LABEL_ROUTE_ADD, sizeof(req), &req, NULL) < 0) + return CMD_ERROR; + + return CMD_SUCCESS; +} + +static cmd_status_t mpls_route_del(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_mpls_label_route_del_req req; + + req = (struct gr_mpls_label_route_del_req) {.missing_ok = true}; + + if (arg_u32(p, "LABEL", &req.in_label) < 0) + return CMD_ERROR; + if (arg_vrf(c, p, "VRF", &req.vrf_id) < 0) + return CMD_ERROR; + + if (gr_api_client_send_recv(c, GR_MPLS_LABEL_ROUTE_DEL, sizeof(req), &req, NULL) < 0) + return CMD_ERROR; + + return CMD_SUCCESS; +} + +static cmd_status_t mpls_route_get(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_mpls_label_route_get_req req; + const struct gr_mpls_label_route *resp; + struct gr_object *o; + void *resp_ptr; + + req = (struct gr_mpls_label_route_get_req) {0}; + resp_ptr = NULL; + + if (arg_u32(p, "LABEL", &req.in_label) < 0) + return CMD_ERROR; + if (arg_vrf(c, p, "VRF", &req.vrf_id) < 0) + return CMD_ERROR; + + if (gr_api_client_send_recv(c, GR_MPLS_LABEL_ROUTE_GET, sizeof(req), &req, &resp_ptr) < 0) + return CMD_ERROR; + + resp = resp_ptr; + o = gr_object_new(NULL); + gr_object_field(o, "vrf", 0, "%s", iface_name_from_id(c, resp->vrf_id)); + gr_object_field(o, "label", 0, "%u", resp->in_label); + gr_object_field(o, "nexthop_id", 0, "%u", resp->nh_id); + gr_object_field(o, "origin", 0, "%s", gr_nh_origin_name(resp->origin)); + gr_object_free(o); + free(resp_ptr); + + return CMD_SUCCESS; +} + +static cmd_status_t mpls_route_list(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_mpls_label_route_list_req req; + const struct gr_mpls_label_route *route; + struct gr_table *table; + uint16_t max_routes; + uint16_t vrf_id; + int ret; + + vrf_id = GR_VRF_ID_UNDEF; + max_routes = 1000; + + if (arg_str(p, "VRF") != NULL && arg_vrf(c, p, "VRF", &vrf_id) < 0) + return CMD_ERROR; + if (arg_u16(p, "MAX", &max_routes) < 0 && errno != ENOENT) + return CMD_ERROR; + + req = (struct gr_mpls_label_route_list_req) { + .vrf_id = vrf_id, + .max_count = max_routes, + }; + + table = gr_table_new(); + gr_table_column(table, "VRF", GR_DISP_LEFT); + gr_table_column(table, "LABEL", GR_DISP_RIGHT); + gr_table_column(table, "NEXTHOP_ID", GR_DISP_RIGHT); + gr_table_column(table, "ORIGIN", GR_DISP_LEFT); + + gr_api_client_stream_foreach (route, ret, c, GR_MPLS_LABEL_ROUTE_LIST, sizeof(req), &req) { + gr_table_cell(table, 0, "%s", iface_name_from_id(c, route->vrf_id)); + gr_table_cell(table, 1, "%u", route->in_label); + gr_table_cell(table, 2, "%u", route->nh_id); + gr_table_cell(table, 3, "%s", gr_nh_origin_name(route->origin)); + + if (gr_table_print_row(table) < 0) + break; + } + + gr_table_free(table); + + if (ret < 0 && errno == EXFULL) { + warnf("more routes not displayed"); + ret = 0; + } + + return ret < 0 ? CMD_ERROR : CMD_SUCCESS; +} + +#define MPLS_CTX(root) CLI_CONTEXT(root, CTX_ARG("mpls", "MPLS label switching.")) +#define MPLS_ROUTE_CTX(root) CLI_CONTEXT(MPLS_CTX(root), CTX_ARG("route", "Label routes.")) + +static int ctx_init(struct ec_node *root) { + int ret; + + ret = CLI_COMMAND( + MPLS_ROUTE_CTX(root), + "add LABEL nexthop id ID [vrf VRF]", + mpls_route_add, + "Add a label route.", + with_help("MPLS label value (0-1048575).", ec_node_uint("LABEL", 0, 1048575, 10)), + with_help("Nexthop user ID.", ec_node_uint("ID", 1, UINT32_MAX - 1, 10)), + with_help("L3 routing domain name.", ec_node_dyn("VRF", complete_vrf_names, NULL)) + ); + if (ret < 0) + return ret; + ret = CLI_COMMAND( + MPLS_ROUTE_CTX(root), + "del LABEL [vrf VRF]", + mpls_route_del, + "Delete a label route.", + with_help("MPLS label value.", ec_node_uint("LABEL", 0, 1048575, 10)), + with_help("L3 routing domain name.", ec_node_dyn("VRF", complete_vrf_names, NULL)) + ); + if (ret < 0) + return ret; + ret = CLI_COMMAND( + MPLS_ROUTE_CTX(root), + "get LABEL [vrf VRF]", + mpls_route_get, + "Get a label route.", + with_help("MPLS label value.", ec_node_uint("LABEL", 0, 1048575, 10)), + with_help("L3 routing domain name.", ec_node_dyn("VRF", complete_vrf_names, NULL)) + ); + if (ret < 0) + return ret; + ret = CLI_COMMAND( + MPLS_ROUTE_CTX(root), + "[show] [(vrf VRF),(max MAX)]", + mpls_route_list, + "Show label routes.", + with_help( + "Max. number of routes to display (default 1000, use 0 for unlimited).", + ec_node_uint("MAX", 0, UINT16_MAX, 10) + ), + with_help("L3 routing domain name.", ec_node_dyn("VRF", complete_vrf_names, NULL)) + ); + if (ret < 0) + return ret; + + return 0; +} + +static void mpls_event_print(uint32_t event, const void *obj) { + const struct gr_mpls_label_route *r = obj; + const char *action; + + switch (event) { + case GR_EVENT_MPLS_ROUTE_ADD: + action = "add"; + break; + case GR_EVENT_MPLS_ROUTE_DEL: + action = "del"; + break; + default: + action = "?"; + break; + } + + printf("mpls route %s: vrf=%s label=%u nh_id=%u origin=%s\n", + action, + iface_name_from_id(NULL, r->vrf_id), + r->in_label, + r->nh_id, + gr_nh_origin_name(r->origin)); +} + +static struct cli_event_printer printer = { + .name = "mpls_route", + .print = mpls_event_print, + .ev_count = 2, + .ev_types = { + GR_EVENT_MPLS_ROUTE_ADD, + GR_EVENT_MPLS_ROUTE_DEL, + }, +}; + +static struct cli_context ctx = { + .name = "mpls", + .init = ctx_init, +}; + +static void __attribute__((constructor, used)) init(void) { + cli_context_register(&ctx); + cli_event_printer_register(&printer); +} diff --git a/modules/mpls/cli/meson.build b/modules/mpls/cli/meson.build new file mode 100644 index 000000000..d28bae09e --- /dev/null +++ b/modules/mpls/cli/meson.build @@ -0,0 +1,8 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +cli_src += files( + 'label.c', + 'nexthop.c', +) +cli_inc += include_directories('.') diff --git a/modules/mpls/cli/nexthop.c b/modules/mpls/cli/nexthop.c new file mode 100644 index 000000000..87fe1f267 --- /dev/null +++ b/modules/mpls/cli/nexthop.c @@ -0,0 +1,199 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "cli.h" +#include "cli_iface.h" +#include "cli_nexthop.h" +#include "display.h" + +#include +#include +#include + +#include + +#include + +static cmd_status_t nh_mpls_add(struct gr_api_client *c, const struct ec_pnode *p) { + struct gr_nexthop_info_mpls *info; + struct gr_nh_add_req *req; + const struct ec_pnode *n; + unsigned long val; + cmd_status_t ret; + const char *str; + size_t len; + + req = NULL; + ret = CMD_ERROR; + + len = sizeof(*req) + sizeof(*info); + req = calloc(1, len); + if (req == NULL) + goto out; + + req->exist_ok = true; + req->nh.type = GR_NH_T_MPLS; + req->nh.origin = GR_NH_ORIGIN_STATIC; + + if (arg_u32(p, "ID", &req->nh.nh_id) < 0 && errno != ENOENT) + goto out; + if (arg_iface(c, p, "IFACE", GR_IFACE_TYPE_UNDEF, &req->nh.iface_id) < 0) + goto out; + + info = (struct gr_nexthop_info_mpls *)req->nh.info; + + switch (arg_ip4(p, "VIA", &info->via.ipv4)) { + case 0: + info->via.af = GR_AF_IP4; + break; + case -EINVAL: + if (arg_ip6(p, "VIA", &info->via.ipv6) < 0) + goto out; + info->via.af = GR_AF_IP6; + break; + default: + errno = EINVAL; + goto out; + } + + if (arg_u8(p, "TTL", &info->ttl) < 0 && errno != ENOENT) + goto out; + + if (arg_str(p, "ipv4") != NULL) + info->payload_af = GR_AF_IP4; + else if (arg_str(p, "ipv6") != NULL) + info->payload_af = GR_AF_IP6; + + n = ec_pnode_find(p, "LABEL"); + if (n != NULL) { + n = ec_pnode_get_parent(n); + if (n == NULL || ec_pnode_len(n) < 1) { + errno = EINVAL; + goto out; + } + if (ec_pnode_len(n) > GR_MPLS_MAX_LABELS) { + errno = E2BIG; + goto out; + } + for (n = ec_pnode_get_first_child(n); n != NULL; n = ec_pnode_next(n)) { + str = ec_strvec_val(ec_pnode_get_strvec(n), 0); + val = strtoul(str, NULL, 10); + if (val > GR_MPLS_LABEL_MAX) { + errno = EINVAL; + goto out; + } + info->labels[info->n_labels++] = (uint32_t)val; + } + } + + if (gr_api_client_send_recv(c, GR_NH_ADD, len, req, NULL) < 0) + goto out; + + ret = CMD_SUCCESS; +out: + free(req); + return ret; +} + +static void add_columns_mpls(struct gr_table *table) { + gr_table_column(table, "VIA", GR_DISP_LEFT); + gr_table_column(table, "LABELS", GR_DISP_STR_ARRAY); + gr_table_column(table, "TTL", GR_DISP_RIGHT); +} + +static void fill_table_mpls(struct gr_table *table, unsigned start_col, const void *nexthop_info) { + const struct gr_nexthop_info_mpls *info = nexthop_info; + char buf[256]; + ssize_t n; + + buf[0] = '\0'; + n = 0; + + if (info->via.af == GR_AF_IP4) + gr_table_cell(table, start_col, IP4_F, &info->via.ipv4); + else if (info->via.af == GR_AF_IP6) + gr_table_cell(table, start_col, IP6_F, &info->via.ipv6); + else + gr_table_cell(table, start_col, "-"); + + for (uint8_t i = 0; i < info->n_labels; i++) { + SAFE_BUF(snprintf, sizeof(buf), "%s%u", i > 0 ? " " : "", info->labels[i]); + if (sizeof(buf) - n < 20) { + SAFE_BUF(snprintf, sizeof(buf), " ... (%u more)", info->n_labels - i - 1); + break; + } + } +err: + if (info->n_labels > 0 && n > 0) + gr_table_cell(table, start_col + 1, "%s", buf); + else + gr_table_cell(table, start_col + 1, "(pop)"); + + if (info->ttl > 0) + gr_table_cell(table, start_col + 2, "%u", info->ttl); + else + gr_table_cell(table, start_col + 2, "-"); +} + +static void fill_object_mpls(struct gr_object *o, const void *nexthop_info) { + const struct gr_nexthop_info_mpls *info = nexthop_info; + + if (info->via.af == GR_AF_IP4) + gr_object_field(o, "via", 0, IP4_F, &info->via.ipv4); + else if (info->via.af == GR_AF_IP6) + gr_object_field(o, "via", 0, IP6_F, &info->via.ipv6); + + if (info->n_labels > 0) { + gr_object_array_open(o, "labels"); + for (uint8_t i = 0; i < info->n_labels; i++) + gr_object_array_item(o, GR_DISP_INT, "%u", info->labels[i]); + gr_object_array_close(o); + } else { + gr_object_field(o, "action", 0, "pop"); + } + + if (info->ttl > 0) + gr_object_field(o, "ttl", GR_DISP_INT, "%u", info->ttl); + if (info->payload_af != GR_AF_UNSPEC) + gr_object_field(o, "payload", 0, "%s", gr_af_name(info->payload_af)); +} + +static struct cli_nexthop_formatter mpls_formatter = { + .name = "mpls", + .type = GR_NH_T_MPLS, + .add_columns = add_columns_mpls, + .fill_table = fill_table_mpls, + .fill_object = fill_object_mpls, +}; + +static int ctx_init(struct ec_node *root) { + int ret; + + ret = CLI_COMMAND( + NEXTHOP_ADD_CTX(root), + "mpls iface IFACE via VIA [labels LABEL+] [(id ID),(ttl TTL),(payload ipv4|ipv6)]", + nh_mpls_add, + "Add an MPLS nexthop.", + with_help("Output interface.", ec_node_dyn("IFACE", complete_iface_names, NULL)), + with_help("Gateway IPv4/6 address.", ec_node_re("VIA", IP_ANY_RE)), + with_help("Output MPLS label (0-1048575).", ec_node_uint("LABEL", 0, 1048575, 10)), + with_help("Nexthop ID.", ec_node_uint("ID", 1, UINT32_MAX - 1, 10)), + with_help("Initial TTL (1-255, 0=copy).", ec_node_uint("TTL", 1, 255, 10)), + with_help("IPv4 payload.", ec_node_str("ipv4", "ipv4")), + with_help("IPv6 payload.", ec_node_str("ipv6", "ipv6")) + ); + if (ret < 0) + return ret; + + return 0; +} + +static struct cli_context ctx = { + .name = "mpls_nexthop", + .init = ctx_init, +}; + +static void __attribute__((constructor, used)) init(void) { + cli_context_register(&ctx); + cli_nexthop_formatter_register(&mpls_formatter); +} diff --git a/modules/mpls/control/label.c b/modules/mpls/control/label.c new file mode 100644 index 000000000..5e74e671e --- /dev/null +++ b/modules/mpls/control/label.c @@ -0,0 +1,192 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "event.h" +#include "module.h" +#include "mpls.h" + +#include + +static struct api_out label_route_add(const void *request, struct api_ctx *) { + const struct gr_mpls_label_route_add_req *req = request; + struct nexthop *nh; + int ret; + + if (req->in_label > GR_MPLS_LABEL_MAX) + return api_out(EINVAL, 0, NULL); + + nh = nexthop_lookup_id(req->nh_id); + if (nh == NULL) + return api_out(ENOENT, 0, NULL); + if (nh->type != GR_NH_T_MPLS && nh->type != GR_NH_T_GROUP) + return api_out(EINVAL, 0, NULL); + + ret = mpls_rib_insert(req->vrf_id, req->in_label, nh, req->origin, req->exist_ok); + + return api_out(-ret, 0, NULL); +} + +static struct api_out label_route_del(const void *request, struct api_ctx *) { + const struct gr_mpls_label_route_del_req *req = request; + int ret; + + ret = mpls_rib_delete(req->vrf_id, req->in_label, req->missing_ok); + + return api_out(-ret, 0, NULL); +} + +static struct api_out label_route_get(const void *request, struct api_ctx *) { + const struct gr_mpls_label_route_get_req *req = request; + struct gr_mpls_label_route *resp; + const struct nexthop *nh; + + nh = mpls_fib_lookup(req->vrf_id, req->in_label); + if (nh == NULL) + return api_out(ENOENT, 0, NULL); + + resp = calloc(1, sizeof(*resp)); + if (resp == NULL) + return api_out(ENOMEM, 0, NULL); + + resp->vrf_id = req->vrf_id; + resp->in_label = req->in_label; + resp->nh_id = nh->nh_id; + resp->origin = nh->origin; + + return api_out(0, sizeof(*resp), resp); +} + +struct label_list_ctx { + struct api_ctx *ctx; + uint16_t max_count; + uint16_t count; +}; + +static int label_route_send(uint16_t vrf_id, uint32_t label, const struct nexthop *nh, void *priv) { + struct label_list_ctx *lctx = priv; + struct gr_mpls_label_route route; + + if (lctx->max_count != 0 && lctx->count >= lctx->max_count) + return errno_set(EXFULL); + + route = (struct gr_mpls_label_route) { + .vrf_id = vrf_id, + .in_label = label, + .nh_id = nh->nh_id, + .origin = nh->origin, + }; + + api_send(lctx->ctx, sizeof(route), &route); + lctx->count++; + + return 0; +} + +static struct api_out label_route_list(const void *request, struct api_ctx *ctx) { + const struct gr_mpls_label_route_list_req *req = request; + struct label_list_ctx lctx; + int ret; + + lctx = (struct label_list_ctx) { + .ctx = ctx, + .max_count = req->max_count, + }; + + ret = mpls_rib_iter(req->vrf_id, label_route_send, &lctx); + if (ret == -EXFULL) + ret = 0; + + return api_out(-ret, 0, NULL); +} + +// label range reservation — simple linked list; small number of entries expected + +struct label_range_entry { + struct gr_mpls_label_range range; + struct label_range_entry *next; +}; + +static struct label_range_entry *label_ranges; + +static struct api_out label_range_add(const void *request, struct api_ctx *) { + const struct gr_mpls_label_range_add_req *req = request; + struct label_range_entry *e; + + if (req->start > req->end || req->end > GR_MPLS_LABEL_MAX) + return api_out(EINVAL, 0, NULL); + + for (struct label_range_entry *ex = label_ranges; ex != NULL; ex = ex->next) { + if (ex->range.vrf_id != req->vrf_id) + continue; + if (ex->range.start == req->start && ex->range.end == req->end) { + if (!req->exist_ok) + return api_out(EEXIST, 0, NULL); + return api_out(0, 0, NULL); + } + } + + e = calloc(1, sizeof(*e)); + if (e == NULL) + return api_out(ENOMEM, 0, NULL); + + e->range.vrf_id = req->vrf_id; + e->range.start = req->start; + e->range.end = req->end; + e->range.origin = req->origin; + e->next = label_ranges; + label_ranges = e; + + return api_out(0, 0, NULL); +} + +static struct api_out label_range_del(const void *request, struct api_ctx *) { + const struct gr_mpls_label_range_del_req *req = request; + struct label_range_entry **pp; + + pp = &label_ranges; + + while (*pp != NULL) { + struct label_range_entry *e = *pp; + if (e->range.vrf_id == req->vrf_id && e->range.start == req->start + && e->range.end == req->end) { + *pp = e->next; + free(e); + return api_out(0, 0, NULL); + } + pp = &e->next; + } + + if (!req->missing_ok) + return api_out(ENOENT, 0, NULL); + return api_out(0, 0, NULL); +} + +static struct api_out label_range_list(const void *request, struct api_ctx *ctx) { + const struct gr_mpls_label_range_list_req *req = request; + + for (struct label_range_entry *e = label_ranges; e != NULL; e = e->next) { + if (req->vrf_id != 0 && e->range.vrf_id != req->vrf_id) + continue; + api_send(ctx, sizeof(e->range), &e->range); + } + + return api_out(0, 0, NULL); +} + +static struct module mpls_module = { + .name = "mpls", + .depends_on = "nexthop", +}; + +RTE_INIT(mpls_constructor) { + module_register(&mpls_module); + api_handler(GR_MPLS_LABEL_ROUTE_ADD, label_route_add); + api_handler(GR_MPLS_LABEL_ROUTE_DEL, label_route_del); + api_handler(GR_MPLS_LABEL_ROUTE_GET, label_route_get); + api_handler(GR_MPLS_LABEL_ROUTE_LIST, label_route_list); + event_serializer(GR_EVENT_MPLS_ROUTE_ADD, NULL); + event_serializer(GR_EVENT_MPLS_ROUTE_DEL, NULL); + api_handler(GR_MPLS_LABEL_RANGE_ADD, label_range_add); + api_handler(GR_MPLS_LABEL_RANGE_DEL, label_range_del); + api_handler(GR_MPLS_LABEL_RANGE_LIST, label_range_list); +} diff --git a/modules/mpls/control/label_table.c b/modules/mpls/control/label_table.c new file mode 100644 index 000000000..3984326f6 --- /dev/null +++ b/modules/mpls/control/label_table.c @@ -0,0 +1,199 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "config.h" +#include "event.h" +#include "log.h" +#include "mpls.h" + +#include + +#include + +LOG_TYPE("mpls"); + +#define MPLS_LFIB_SIZE (GR_MPLS_LABEL_MAX + 1) + +static struct nexthop **get_lfib(uint16_t vrf_id) { + struct nexthop **lfib; + struct iface *iface; + + iface = get_vrf_iface(vrf_id); + if (iface == NULL) + return NULL; + lfib = iface_info_vrf(iface)->fib_mpls; + if (lfib == NULL) + return errno_set_null(ENONET); + return lfib; +} + +const struct nexthop *mpls_fib_lookup(uint16_t vrf_id, uint32_t label) { + struct nexthop **lfib; + + lfib = get_lfib(vrf_id); + if (lfib == NULL || label > GR_MPLS_LABEL_MAX) + return NULL; + + return lfib[label]; +} + +int mpls_rib_insert( + uint16_t vrf_id, + uint32_t label, + struct nexthop *nh, + gr_nh_origin_t origin, + bool exist_ok +) { + struct nexthop *existing; + struct nexthop **lfib; + + lfib = get_lfib(vrf_id); + + if (lfib == NULL) + return -errno; + if (label > GR_MPLS_LABEL_MAX) + return errno_set(EINVAL); + + existing = lfib[label]; + if (existing != NULL) { + if (!exist_ok) + return errno_set(EEXIST); + if (existing == nh) + return 0; + } + + nexthop_incref(nh); + lfib[label] = nh; + + if (origin != GR_NH_ORIGIN_INTERNAL) { + event_push( + GR_EVENT_MPLS_ROUTE_ADD, + &(const struct gr_mpls_label_route) { + .vrf_id = vrf_id, + .in_label = label, + .nh_id = nh->nh_id, + .origin = origin, + } + ); + } + + if (existing != NULL) + nexthop_decref(existing); + + return 0; +} + +int mpls_rib_delete(uint16_t vrf_id, uint32_t label, bool missing_ok) { + struct nexthop **lfib; + struct nexthop *nh; + + lfib = get_lfib(vrf_id); + + if (lfib == NULL) + return -errno; + if (label > GR_MPLS_LABEL_MAX) + return errno_set(EINVAL); + + nh = lfib[label]; + if (nh == NULL) { + if (missing_ok) + return 0; + return errno_set(ENOENT); + } + + lfib[label] = NULL; + + if (nh->origin != GR_NH_ORIGIN_INTERNAL) { + event_push( + GR_EVENT_MPLS_ROUTE_DEL, + &(const struct gr_mpls_label_route) { + .vrf_id = vrf_id, + .in_label = label, + .nh_id = nh->nh_id, + .origin = nh->origin, + } + ); + } + + nexthop_decref(nh); + + return 0; +} + +int mpls_rib_iter(uint16_t vrf_id, mpls_rib_iter_cb cb, void *priv) { + struct nexthop **lfib; + struct iface *iface; + int ret; + + if (vrf_id != GR_VRF_ID_UNDEF) { + lfib = get_lfib(vrf_id); + if (lfib == NULL) + return -errno; + for (uint32_t i = 0; i < MPLS_LFIB_SIZE; i++) { + if (lfib[i] == NULL) + continue; + ret = cb(vrf_id, i, lfib[i], priv); + if (ret < 0) + return ret; + } + } else { + for (uint16_t v = 1; v < gr_config.max_ifaces; v++) { + iface = iface_from_id(v); + if (iface == NULL || iface->type != GR_IFACE_TYPE_VRF) + continue; + lfib = iface_info_vrf(iface)->fib_mpls; + if (lfib == NULL) + continue; + for (uint32_t i = 0; i < MPLS_LFIB_SIZE; i++) { + if (lfib[i] == NULL) + continue; + ret = cb(v, i, lfib[i], priv); + if (ret < 0) + return ret; + } + } + } + return 0; +} + +static int mpls_fib_init(struct iface *vrf) { + struct nexthop **lfib; + + lfib = rte_zmalloc("mpls_lfib", MPLS_LFIB_SIZE * sizeof(*lfib), RTE_CACHE_LINE_SIZE); + if (lfib == NULL) + return errno_log(rte_errno, "rte_zmalloc(mpls_lfib)"); + + iface_info_vrf(vrf)->fib_mpls = lfib; + + return 0; +} + +static int mpls_fib_reconfig(struct iface *) { + return 0; +} + +static void mpls_fib_fini(struct iface *vrf) { + struct nexthop **lfib; + + lfib = iface_info_vrf(vrf)->fib_mpls; + if (lfib == NULL) + return; + + for (uint32_t i = 0; i < MPLS_LFIB_SIZE; i++) { + if (lfib[i] != NULL) + nexthop_decref(lfib[i]); + } + + rte_free(lfib); + iface_info_vrf(vrf)->fib_mpls = NULL; +} + +static const struct vrf_fib_ops mpls_fib_ops = { + .init = mpls_fib_init, + .reconfig = mpls_fib_reconfig, + .fini = mpls_fib_fini, +}; + +RTE_INIT(mpls_label_table_init) { + vrf_fib_ops_register(GR_AF_MPLS, &mpls_fib_ops); +} diff --git a/modules/mpls/control/meson.build b/modules/mpls/control/meson.build new file mode 100644 index 000000000..2463eeed6 --- /dev/null +++ b/modules/mpls/control/meson.build @@ -0,0 +1,20 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +src += files( + 'label.c', + 'label_table.c', + 'nexthop.c', + 'resolve.c', +) +inc += include_directories('.') + +tests += [ + { + 'sources': files('mpls_test.c', 'label_table.c', 'nexthop.c'), + 'link_args': [ + '-Wl,--wrap=rte_zmalloc', + '-Wl,--wrap=rte_free', + ], + }, +] diff --git a/modules/mpls/control/mpls.h b/modules/mpls/control/mpls.h new file mode 100644 index 000000000..bd1fd8898 --- /dev/null +++ b/modules/mpls/control/mpls.h @@ -0,0 +1,26 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#pragma once + +#include "nexthop.h" +#include "vrf.h" + +#include + +GR_NH_TYPE_INFO(GR_NH_T_MPLS, nexthop_info_mpls, { + uint8_t n_labels; + uint8_t ttl; + addr_family_t payload_af; + uint32_t labels[GR_MPLS_MAX_LABELS]; + struct nexthop *via_nh; +}); + +// Look up a label in the per-VRF LFIB. Called from the datapath. +const struct nexthop *mpls_fib_lookup(uint16_t vrf_id, uint32_t label); + +int mpls_rib_insert(uint16_t, uint32_t, struct nexthop *, gr_nh_origin_t, bool exist_ok); +int mpls_rib_delete(uint16_t vrf_id, uint32_t label, bool missing_ok); + +typedef int (*mpls_rib_iter_cb)(uint16_t, uint32_t, const struct nexthop *, void *); +int mpls_rib_iter(uint16_t vrf_id, mpls_rib_iter_cb cb, void *priv); diff --git a/modules/mpls/control/mpls_test.c b/modules/mpls/control/mpls_test.c new file mode 100644 index 000000000..58aa33604 --- /dev/null +++ b/modules/mpls/control/mpls_test.c @@ -0,0 +1,248 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "_cmocka.h" +#include "config.h" +#include "event.h" +#include "log.h" +#include "module.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include + +// Global variables declared extern in log.h; defined here for the test binary. +int gr_rte_log_type; +struct log_types log_types = STAILQ_HEAD_INITIALIZER(log_types); +struct gr_config gr_config; + +// Stubs for functions used by label_table.c and nexthop.c that are not under +// test. The prototypes are provided by the included headers above. +void module_register(struct module *) { } +void event_push(uint32_t, const void *) { } +void event_subscribe(uint32_t, event_sub_cb_t) { } +void nexthop_incref(struct nexthop *) { } +void nexthop_decref(struct nexthop *) { } +void nexthop_type_ops_register(gr_nh_type_t, const struct nexthop_type_ops *) { } +void vrf_fib_ops_register(addr_family_t, const struct vrf_fib_ops *) { } +struct nexthop *nexthop_new(const struct gr_nexthop_base *, const void *) { + return NULL; +} +struct nexthop *nexthop_lookup_l3(addr_family_t, uint16_t, uint16_t, const void *) { + return NULL; +} +void nexthop_iter(nh_iter_cb_t, void *) { } +struct iface *iface_from_id(uint16_t) { + return NULL; +} + +// rte_zmalloc wrapped to avoid DPDK EAL init in tests; forward declarations +// satisfy -Wmissing-prototypes before the definitions. +void *__wrap_rte_zmalloc(const char *, size_t size, unsigned /*align*/); +void *__wrap_rte_zmalloc(const char *, size_t size, unsigned /*align*/) { + return calloc(1, size); +} +void __wrap_rte_free(void *ptr); +void __wrap_rte_free(void *ptr) { + free(ptr); +} + +// Fake VRF iface: the flexible array trick lets iface_info_vrf(&fake_vrf.iface) +// point directly at the embedded vrf member. +static struct { + struct iface iface; + struct iface_info_vrf vrf; +} fake_vrf; + +struct iface *get_vrf_iface(uint16_t) { + return &fake_vrf.iface; +} + +static void test_label_encode_decode(void **) { + static const uint32_t labels[] = {0, 15, 16, 100, 0xFFFFF}; + for (size_t i = 0; i < sizeof(labels) / sizeof(labels[0]); i++) { + struct rte_mpls_hdr h = {}; + h.bs = 1; + h.tc = 7; + mpls_hdr_set_label(&h, labels[i]); + assert_int_equal(mpls_hdr_get_label(&h), labels[i]); + assert_int_equal(h.bs, 1); + assert_int_equal(h.tc, 7); + } +} + +static void test_label_split(void **) { + struct rte_mpls_hdr h = {}; + mpls_hdr_set_label(&h, 0x12345); + assert_int_equal(h.tag_lsb, 0x5); + assert_int_equal(mpls_hdr_get_label(&h), 0x12345); + + mpls_hdr_set_label(&h, 0); + assert_int_equal(h.tag_lsb, 0); + assert_int_equal(mpls_hdr_get_label(&h), 0); +} + +static int lfib_setup(void **) { + size_t sz; + + memset(&fake_vrf, 0, sizeof(fake_vrf)); + fake_vrf.iface.type = GR_IFACE_TYPE_VRF; + sz = (GR_MPLS_LABEL_MAX + 1) * sizeof(struct nexthop *); + iface_info_vrf(&fake_vrf.iface)->fib_mpls = calloc(1, sz); + return 0; +} + +static int lfib_teardown(void **) { + free(iface_info_vrf(&fake_vrf.iface)->fib_mpls); + iface_info_vrf(&fake_vrf.iface)->fib_mpls = NULL; + return 0; +} + +static void test_lfib_insert_lookup(void **) { + struct nexthop nh = {}; + nh.origin = GR_NH_ORIGIN_ZSTATIC; + + assert_int_equal(mpls_rib_insert(1, 100, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_ptr_equal(mpls_fib_lookup(1, 100), &nh); + assert_null(mpls_fib_lookup(1, 101)); +} + +static void test_lfib_insert_duplicate(void **) { + struct nexthop nh = {}; + nh.origin = GR_NH_ORIGIN_ZSTATIC; + + assert_int_equal(mpls_rib_insert(1, 200, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_int_equal(mpls_rib_insert(1, 200, &nh, GR_NH_ORIGIN_ZSTATIC, true), 0); + assert_int_equal(mpls_rib_insert(1, 200, &nh, GR_NH_ORIGIN_ZSTATIC, false), -EEXIST); +} + +static void test_lfib_delete(void **) { + struct nexthop nh = {}; + nh.origin = GR_NH_ORIGIN_ZSTATIC; + + assert_int_equal(mpls_rib_insert(1, 300, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_int_equal(mpls_rib_delete(1, 300, false), 0); + assert_null(mpls_fib_lookup(1, 300)); + assert_int_equal(mpls_rib_delete(1, 300, true), 0); + assert_int_equal(mpls_rib_delete(1, 300, false), -ENOENT); +} + +static int g_iter_count; +static int count_cb(uint16_t, uint32_t, const struct nexthop *, void *) { + g_iter_count++; + return 0; +} + +static void test_lfib_iter(void **) { + struct nexthop nh = {}; + nh.origin = GR_NH_ORIGIN_ZSTATIC; + + assert_int_equal(mpls_rib_insert(1, 400, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_int_equal(mpls_rib_insert(1, 401, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + assert_int_equal(mpls_rib_insert(1, 402, &nh, GR_NH_ORIGIN_ZSTATIC, false), 0); + + g_iter_count = 0; + assert_int_equal(mpls_rib_iter(1, count_cb, NULL), 0); + assert_int_equal(g_iter_count, 3); +} + +static void test_lfib_label_invalid(void **) { + struct nexthop nh = {}; + uint32_t bad; + + bad = GR_MPLS_LABEL_MAX + 1; + + assert_int_equal(mpls_rib_insert(1, bad, &nh, GR_NH_ORIGIN_ZSTATIC, false), -EINVAL); + assert_null(mpls_fib_lookup(1, bad)); + assert_int_equal(mpls_rib_delete(1, bad, false), -EINVAL); +} + +extern bool mpls_nh_equal_test(const struct nexthop *, const struct nexthop *); + +static void test_nh_equal_identical(void **) { + struct nexthop a = {}, b = {}; + a.type = GR_NH_T_MPLS; + b.type = GR_NH_T_MPLS; + struct nexthop_info_mpls *ma = nexthop_info_mpls(&a); + struct nexthop_info_mpls *mb = nexthop_info_mpls(&b); + + ma->n_labels = 1; + ma->labels[0] = 100; + ma->payload_af = GR_AF_UNSPEC; + ma->via_nh = NULL; + memcpy(mb, ma, sizeof(*mb)); + + assert_true(mpls_nh_equal_test(&a, &b)); +} + +static void test_nh_equal_different_n_labels(void **) { + struct nexthop a = {}, b = {}; + a.type = GR_NH_T_MPLS; + b.type = GR_NH_T_MPLS; + struct nexthop_info_mpls *ma = nexthop_info_mpls(&a); + struct nexthop_info_mpls *mb = nexthop_info_mpls(&b); + + ma->n_labels = 1; + ma->labels[0] = 100; + mb->n_labels = 2; + mb->labels[0] = 100; + mb->labels[1] = 200; + + assert_false(mpls_nh_equal_test(&a, &b)); +} + +static void test_nh_equal_different_label(void **) { + struct nexthop a = {}, b = {}; + a.type = GR_NH_T_MPLS; + b.type = GR_NH_T_MPLS; + struct nexthop_info_mpls *ma = nexthop_info_mpls(&a); + struct nexthop_info_mpls *mb = nexthop_info_mpls(&b); + + ma->n_labels = 1; + ma->labels[0] = 100; + mb->n_labels = 1; + mb->labels[0] = 200; + + assert_false(mpls_nh_equal_test(&a, &b)); +} + +static void test_nh_equal_different_payload_af(void **) { + struct nexthop a = {}, b = {}; + a.type = GR_NH_T_MPLS; + b.type = GR_NH_T_MPLS; + struct nexthop_info_mpls *ma = nexthop_info_mpls(&a); + struct nexthop_info_mpls *mb = nexthop_info_mpls(&b); + + ma->n_labels = 1; + ma->labels[0] = 100; + ma->payload_af = GR_AF_IP4; + mb->n_labels = 1; + mb->labels[0] = 100; + mb->payload_af = GR_AF_IP6; + + assert_false(mpls_nh_equal_test(&a, &b)); +} + +int main(void) { + const struct CMUnitTest tests[] = { + // Group A: header encoding + cmocka_unit_test(test_label_encode_decode), + cmocka_unit_test(test_label_split), + // Group B: LFIB operations + cmocka_unit_test_setup_teardown(test_lfib_insert_lookup, lfib_setup, lfib_teardown), + cmocka_unit_test_setup_teardown( + test_lfib_insert_duplicate, lfib_setup, lfib_teardown + ), + cmocka_unit_test_setup_teardown(test_lfib_delete, lfib_setup, lfib_teardown), + cmocka_unit_test_setup_teardown(test_lfib_iter, lfib_setup, lfib_teardown), + cmocka_unit_test_setup_teardown(test_lfib_label_invalid, lfib_setup, lfib_teardown), + // Group C: nexthop equality + cmocka_unit_test(test_nh_equal_identical), + cmocka_unit_test(test_nh_equal_different_n_labels), + cmocka_unit_test(test_nh_equal_different_label), + cmocka_unit_test(test_nh_equal_different_payload_af), + }; + return cmocka_run_group_tests(tests, NULL, NULL); +} diff --git a/modules/mpls/control/nexthop.c b/modules/mpls/control/nexthop.c new file mode 100644 index 000000000..35c9ad1ec --- /dev/null +++ b/modules/mpls/control/nexthop.c @@ -0,0 +1,150 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "mpls.h" + +#include + +static int mpls_nh_import_info(struct nexthop *nh, const void *info) { + struct nexthop_info_mpls *priv = nexthop_info_mpls(nh); + const struct gr_nexthop_info_mpls *pub = info; + struct gr_nexthop_info_l3 l3_info; + struct nexthop *via_nh, *old_via; + + if (pub->n_labels > GR_MPLS_MAX_LABELS) + return errno_set(EINVAL); + + for (uint8_t i = 0; i < pub->n_labels; i++) { + if (pub->labels[i] > GR_MPLS_LABEL_MAX) + return errno_set(EINVAL); + if (pub->labels[i] == GR_MPLS_LABEL_IMPLICIT_NULL) + return errno_set(EINVAL); + } + + switch (pub->payload_af) { + case GR_AF_UNSPEC: + case GR_AF_IP4: + case GR_AF_IP6: + break; + default: + return errno_set(EINVAL); + } + + if (pub->via.af != GR_AF_IP4 && pub->via.af != GR_AF_IP6) + return errno_set(EAFNOSUPPORT); + + via_nh = nexthop_lookup_l3(pub->via.af, nh->vrf_id, nh->iface_id, &pub->via.addr); + if (via_nh == NULL) { + l3_info = (struct gr_nexthop_info_l3) {.af = pub->via.af}; + if (pub->via.af == GR_AF_IP4) + l3_info.ipv4 = pub->via.ipv4; + else + l3_info.ipv6 = pub->via.ipv6; + + via_nh = nexthop_new( + &(struct gr_nexthop_base) { + .type = GR_NH_T_L3, + .iface_id = nh->iface_id, + .vrf_id = nh->vrf_id, + .origin = GR_NH_ORIGIN_INTERNAL, + }, + &l3_info + ); + if (via_nh == NULL) + return -errno; + } else { + nexthop_incref(via_nh); + } + + priv->n_labels = pub->n_labels; + priv->ttl = pub->ttl; + priv->payload_af = pub->payload_af; + memcpy(priv->labels, pub->labels, pub->n_labels * sizeof(pub->labels[0])); + + old_via = priv->via_nh; + priv->via_nh = via_nh; + if (old_via != NULL) + nexthop_decref(old_via); + + return 0; +} + +static void mpls_nh_free(struct nexthop *nh) { + struct nexthop_info_mpls *priv = nexthop_info_mpls(nh); + if (priv->via_nh != NULL) + nexthop_decref(priv->via_nh); + priv->via_nh = NULL; +} + +static void mpls_nh_remove_via_cb(struct nexthop *nh, void *dying) { + if (nh->type != GR_NH_T_MPLS) + return; + struct nexthop_info_mpls *priv = nexthop_info_mpls(nh); + if (priv->via_nh == dying) + priv->via_nh = NULL; +} + +static void mpls_nh_remove_references(struct nexthop *dying) { + nexthop_iter(mpls_nh_remove_via_cb, dying); +} + +static bool mpls_nh_equal(const struct nexthop *a, const struct nexthop *b) { + const struct nexthop_info_mpls *ma = nexthop_info_mpls(a); + const struct nexthop_info_mpls *mb = nexthop_info_mpls(b); + + if (ma->n_labels != mb->n_labels) + return false; + if (ma->payload_af != mb->payload_af) + return false; + if (ma->via_nh != mb->via_nh) + return false; + return memcmp(ma->labels, mb->labels, ma->n_labels * sizeof(ma->labels[0])) == 0; +} + +static struct gr_nexthop *mpls_nh_to_api(const struct nexthop *nh, size_t *len) { + const struct nexthop_info_mpls *priv = nexthop_info_mpls(nh); + struct gr_nexthop_info_mpls *pub_info; + struct gr_nexthop *pub; + + *len = sizeof(*pub) + sizeof(*pub_info); + pub = calloc(1, *len); + if (pub == NULL) + return errno_set_null(ENOMEM); + + pub->base = nh->base; + pub_info = (struct gr_nexthop_info_mpls *)pub->info; + pub_info->n_labels = priv->n_labels; + pub_info->ttl = priv->ttl; + pub_info->payload_af = priv->payload_af; + memcpy(pub_info->labels, priv->labels, priv->n_labels * sizeof(priv->labels[0])); + + if (priv->via_nh != NULL) { + const struct nexthop_info_l3 *l3 = nexthop_info_l3(priv->via_nh); + pub_info->via.af = l3->af; + if (l3->af == GR_AF_IP4) + pub_info->via.ipv4 = l3->ipv4; + else if (l3->af == GR_AF_IP6) + pub_info->via.ipv6 = l3->ipv6; + } + + return pub; +} + +static struct nexthop_type_ops mpls_nh_ops = { + .import_info = mpls_nh_import_info, + .free = mpls_nh_free, + .remove_references = mpls_nh_remove_references, + .equal = mpls_nh_equal, + .to_api = mpls_nh_to_api, +}; + +RTE_INIT(mpls_nexthop_init) { + nexthop_type_ops_register(GR_NH_T_MPLS, &mpls_nh_ops); +} + +#ifdef __GROUT_UNIT_TEST__ +bool mpls_nh_equal_test(const struct nexthop *a, const struct nexthop *b); +bool mpls_nh_equal_test(const struct nexthop *a, const struct nexthop *b) { + return mpls_nh_equal(a, b); +} +#endif diff --git a/modules/mpls/control/resolve.c b/modules/mpls/control/resolve.c new file mode 100644 index 000000000..3fd5da996 --- /dev/null +++ b/modules/mpls/control/resolve.c @@ -0,0 +1,98 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "config.h" +#include "l3.h" +#include "log.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +LOG_TYPE("mpls"); + +static void mpls_resolve_cb(void *obj, uintptr_t, const struct control_queue_drain *drain) { + struct nexthop_info_l3 *l3; + struct rte_mbuf *m = obj; + struct nexthop *nh; + + nh = (struct nexthop *)l3_mbuf_data(m)->nh; + + if (drain != NULL) { + switch (drain->event) { + case GR_EVENT_IFACE_REMOVE: + if (mbuf_data(m)->iface == drain->obj) + goto free; + break; + case GR_EVENT_NEXTHOP_DELETE: + if (nh == drain->obj) + goto free; + if (mpls_hold_mbuf_data(m)->mpls_nh == drain->obj) + goto free; + break; + } + } + + l3 = nexthop_info_l3(nh); + + if (l3->state == GR_NH_S_REACHABLE) { + if (mpls_resubmit_cb(m, nh) < 0) + goto free; + return; + } + + if (l3->held_pkts < nh_conf.max_held_pkts) { + queue_mbuf_data(m)->next = NULL; + if (l3->held_pkts_head == NULL) + l3->held_pkts_head = m; + else + queue_mbuf_data(l3->held_pkts_tail)->next = m; + l3->held_pkts_tail = m; + l3->held_pkts++; + if (l3->state != GR_NH_S_PENDING) { + const struct nexthop_af_ops *ops = nexthop_af_ops_from_nh(nh); + if (ops != NULL) + ops->solicit(nh); + l3->state = GR_NH_S_PENDING; + } + return; + } + +free: + rte_pktmbuf_free(m); +} + +static int mpls_solicit(struct nexthop *nh) { + const struct nexthop_af_ops *ops = nexthop_af_ops_from_nh(nh); + if (ops != NULL) + return ops->solicit(nh); + return errno_set(ENOTSUP); +} + +static void mpls_rib_cleanup(struct nexthop *nh, bool) { + for (uint16_t v = 1; v < gr_config.max_ifaces; v++) { + struct iface *iface = iface_from_id(v); + if (iface == NULL || iface->type != GR_IFACE_TYPE_VRF) + continue; + struct nexthop **lfib = iface_info_vrf(iface)->fib_mpls; + if (lfib == NULL) + continue; + for (uint32_t i = 0; i <= GR_MPLS_LABEL_MAX; i++) { + if (lfib[i] == nh) { + lfib[i] = NULL; + nexthop_decref(nh); + } + } + } +} + +static struct nexthop_af_ops mpls_af_ops = { + .resolve = mpls_resolve_cb, + .solicit = mpls_solicit, + .resubmit = mpls_resubmit_cb, + .cleanup_routes = mpls_rib_cleanup, +}; + +RTE_INIT(mpls_resolve_constructor) { + nexthop_af_ops_register(GR_AF_MPLS, &mpls_af_ops); +} diff --git a/modules/mpls/datapath/meson.build b/modules/mpls/datapath/meson.build new file mode 100644 index 000000000..c946118f6 --- /dev/null +++ b/modules/mpls/datapath/meson.build @@ -0,0 +1,10 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +src += files( + 'mpls_error.c', + 'mpls_input.c', + 'mpls_output.c', + 'mpls_push.c', +) +inc += include_directories('.') diff --git a/modules/mpls/datapath/mpls_datapath.h b/modules/mpls/datapath/mpls_datapath.h new file mode 100644 index 000000000..aa99bb94f --- /dev/null +++ b/modules/mpls/datapath/mpls_datapath.h @@ -0,0 +1,26 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#pragma once + +#include "mbuf.h" +#include "nexthop.h" + +#include +#include + +static inline uint32_t mpls_hdr_get_label(const struct rte_mpls_hdr *h) { + return (rte_be_to_cpu_16(h->tag_msb) << 4) | h->tag_lsb; +} + +static inline void mpls_hdr_set_label(struct rte_mpls_hdr *h, uint32_t label) { + h->tag_msb = rte_cpu_to_be_16(label >> 4); + h->tag_lsb = label & 0xf; +} + +GR_MBUF_PRIV_DATA_TYPE(mpls_hold_mbuf_data, { + const struct nexthop *nh; + const struct nexthop *mpls_nh; +}); + +int mpls_resubmit_cb(struct rte_mbuf *, struct nexthop *); diff --git a/modules/mpls/datapath/mpls_error.c b/modules/mpls/datapath/mpls_error.c new file mode 100644 index 000000000..2ca0aa6ac --- /dev/null +++ b/modules/mpls/datapath/mpls_error.c @@ -0,0 +1,425 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "graph.h" +#include "icmp6.h" +#include "ip4.h" +#include "ip4_datapath.h" +#include "ip6.h" +#include "ip6_datapath.h" +#include "l3.h" +#include "mbuf.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include +#include +#include +#include +#include + +#include + +enum { + ICMP_OUTPUT = 0, + NO_HEADROOM, + NO_IP, + EDGE_COUNT, +}; + +static uint16_t mpls_effective_mtu(struct rte_mbuf *mbuf) { + const struct nexthop *nh = l3_mbuf_data(mbuf)->nh; + const struct nexthop_info_mpls *info = nexthop_info_mpls(nh); + const struct iface *out_iface = iface_from_id(info->via_nh->iface_id); + if (out_iface == NULL) + return 0; + return out_iface->mtu - info->n_labels * sizeof(struct rte_mpls_hdr); +} + +static uint16_t mpls_frag_needed_process( + struct rte_graph *graph, + struct rte_node *node, + void **objs, + uint16_t nb_objs +) { + struct ip_local_mbuf_data *ip_data; + const struct nexthop_info_l3 *l3; + const struct nexthop *nh, *local; + const struct iface *in_iface; + struct rte_icmp_hdr *icmp; + struct rte_ipv4_hdr *ip; + uint16_t effective_mtu; + struct rte_mbuf *mbuf; + ip4_addr_t src, dst; + rte_edge_t edge; + unsigned len; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + effective_mtu = mpls_effective_mtu(mbuf); + + ip = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + src = ip->src_addr; + // RFC 792: IP header + 64 bits of original datagram + len = rte_ipv4_hdr_len(ip) + 8; + rte_pktmbuf_trim(mbuf, rte_pktmbuf_pkt_len(mbuf) - len); + + icmp = gr_mbuf_prepend(mbuf, icmp); + if (unlikely(icmp == NULL)) { + edge = NO_HEADROOM; + goto next; + } + + in_iface = mbuf_data(mbuf)->iface; + if (in_iface == NULL || (nh = fib4_lookup(in_iface->vrf_id, src, 0)) == NULL) { + edge = NO_IP; + goto next; + } + if (nh->type == GR_NH_T_L3) { + l3 = nexthop_info_l3(nh); + dst = l3->ipv4; + } else { + dst = src; + } + if ((local = addr4_get_preferred(nh->iface_id, dst)) == NULL) { + edge = NO_IP; + goto next; + } + + icmp->icmp_type = RTE_ICMP_TYPE_DEST_UNREACHABLE; + icmp->icmp_code = RTE_ICMP_CODE_UNREACH_FRAG; + icmp->icmp_cksum = 0; + icmp->icmp_ident = 0; + icmp->icmp_seq_nb = rte_cpu_to_be_16(effective_mtu); + + l3 = nexthop_info_l3(local); + ip_data = ip_local_mbuf_data(mbuf); + ip_data->src = l3->ipv4; + ip_data->dst = src; + ip_data->vrf_id = in_iface->vrf_id; + ip_data->len = rte_pktmbuf_pkt_len(mbuf); + ip_data->proto = IPPROTO_ICMP; + + edge = ICMP_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static uint16_t mpls_pkt_too_big_process( + struct rte_graph *graph, + struct rte_node *node, + void **objs, + uint16_t nb_objs +) { + struct icmp6_err_pkt_too_big *ptb; + const struct nexthop_info_l3 *l3; + struct ip6_local_mbuf_data *d; + const struct iface *in_iface; + const struct nexthop *local; + struct rte_ipv6_hdr *ip6; + uint16_t effective_mtu; + struct rte_mbuf *mbuf; + struct icmp6 *icmp6; + rte_edge_t edge; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + effective_mtu = mpls_effective_mtu(mbuf); + + ip6 = rte_pktmbuf_mtod(mbuf, struct rte_ipv6_hdr *); + + // RFC 4443: as much of the invoking packet as possible without + // the ICMPv6 packet exceeding the minimum IPv6 MTU (1280) + if (rte_pktmbuf_pkt_len(mbuf) > RTE_IPV6_MIN_MTU) + rte_pktmbuf_trim(mbuf, rte_pktmbuf_pkt_len(mbuf) - RTE_IPV6_MIN_MTU); + + ptb = gr_mbuf_prepend(mbuf, ptb); + if (unlikely(ptb == NULL)) { + edge = NO_HEADROOM; + goto next; + } + ptb->mtu = rte_cpu_to_be_32(effective_mtu); + + icmp6 = gr_mbuf_prepend(mbuf, icmp6); + if (unlikely(icmp6 == NULL)) { + edge = NO_HEADROOM; + goto next; + } + icmp6->type = ICMP6_ERR_PKT_TOO_BIG; + icmp6->code = 0; + + in_iface = mbuf_data(mbuf)->iface; + if (in_iface == NULL) { + edge = NO_IP; + goto next; + } + if ((local = addr6_get_preferred(in_iface->id, &ip6->src_addr)) == NULL) { + edge = NO_IP; + goto next; + } + + l3 = nexthop_info_l3(local); + d = ip6_local_mbuf_data(mbuf); + d->src = l3->ipv6; + d->dst = ip6->src_addr; + d->len = rte_pktmbuf_pkt_len(mbuf); + d->iface = in_iface; + + edge = ICMP_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static bool mpls_strip_label_stack(struct rte_mbuf *mbuf) { + struct rte_mpls_hdr *m; + uint32_t depth = 0; + + m = rte_pktmbuf_mtod(mbuf, struct rte_mpls_hdr *); + + while (depth < GR_MPLS_MAX_STACK_DEPTH) { + if (m->bs) { + rte_pktmbuf_adj(mbuf, (depth + 1) * sizeof(*m)); + return true; + } + m++; + depth++; + } + return false; +} + +static uint16_t mpls_output_frag_needed_process( + struct rte_graph *graph, + struct rte_node *node, + void **objs, + uint16_t nb_objs +) { + struct ip_local_mbuf_data *ip_data; + const struct nexthop_info_l3 *l3; + const struct nexthop *nh, *local; + const struct iface *in_iface; + struct rte_icmp_hdr *icmp; + struct rte_ipv4_hdr *ip; + uint16_t effective_mtu; + struct rte_mbuf *mbuf; + ip4_addr_t src, dst; + rte_edge_t edge; + unsigned len; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + effective_mtu = mpls_effective_mtu(mbuf); + + if (!mpls_strip_label_stack(mbuf)) { + edge = NO_IP; + goto next; + } + + ip = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + src = ip->src_addr; + len = rte_ipv4_hdr_len(ip) + 8; + rte_pktmbuf_trim(mbuf, rte_pktmbuf_pkt_len(mbuf) - len); + + icmp = gr_mbuf_prepend(mbuf, icmp); + if (unlikely(icmp == NULL)) { + edge = NO_HEADROOM; + goto next; + } + + in_iface = mbuf_data(mbuf)->iface; + if (in_iface == NULL || (nh = fib4_lookup(in_iface->vrf_id, src, 0)) == NULL) { + edge = NO_IP; + goto next; + } + if (nh->type == GR_NH_T_L3) { + l3 = nexthop_info_l3(nh); + dst = l3->ipv4; + } else { + dst = src; + } + if ((local = addr4_get_preferred(nh->iface_id, dst)) == NULL) { + edge = NO_IP; + goto next; + } + + icmp->icmp_type = RTE_ICMP_TYPE_DEST_UNREACHABLE; + icmp->icmp_code = RTE_ICMP_CODE_UNREACH_FRAG; + icmp->icmp_cksum = 0; + icmp->icmp_ident = 0; + icmp->icmp_seq_nb = rte_cpu_to_be_16(effective_mtu); + + l3 = nexthop_info_l3(local); + ip_data = ip_local_mbuf_data(mbuf); + ip_data->src = l3->ipv4; + ip_data->dst = src; + ip_data->vrf_id = in_iface->vrf_id; + ip_data->len = rte_pktmbuf_pkt_len(mbuf); + ip_data->proto = IPPROTO_ICMP; + + edge = ICMP_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static uint16_t mpls_output_pkt_too_big_process( + struct rte_graph *graph, + struct rte_node *node, + void **objs, + uint16_t nb_objs +) { + struct icmp6_err_pkt_too_big *ptb; + const struct nexthop_info_l3 *l3; + struct ip6_local_mbuf_data *d; + const struct iface *in_iface; + const struct nexthop *local; + struct rte_ipv6_hdr *ip6; + uint16_t effective_mtu; + struct rte_mbuf *mbuf; + struct icmp6 *icmp6; + rte_edge_t edge; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + effective_mtu = mpls_effective_mtu(mbuf); + + if (!mpls_strip_label_stack(mbuf)) { + edge = NO_IP; + goto next; + } + + ip6 = rte_pktmbuf_mtod(mbuf, struct rte_ipv6_hdr *); + + if (rte_pktmbuf_pkt_len(mbuf) > RTE_IPV6_MIN_MTU) + rte_pktmbuf_trim(mbuf, rte_pktmbuf_pkt_len(mbuf) - RTE_IPV6_MIN_MTU); + + ptb = gr_mbuf_prepend(mbuf, ptb); + if (unlikely(ptb == NULL)) { + edge = NO_HEADROOM; + goto next; + } + ptb->mtu = rte_cpu_to_be_32(effective_mtu); + + icmp6 = gr_mbuf_prepend(mbuf, icmp6); + if (unlikely(icmp6 == NULL)) { + edge = NO_HEADROOM; + goto next; + } + icmp6->type = ICMP6_ERR_PKT_TOO_BIG; + icmp6->code = 0; + + in_iface = mbuf_data(mbuf)->iface; + if (in_iface == NULL) { + edge = NO_IP; + goto next; + } + if ((local = addr6_get_preferred(in_iface->id, &ip6->src_addr)) == NULL) { + edge = NO_IP; + goto next; + } + + l3 = nexthop_info_l3(local); + d = ip6_local_mbuf_data(mbuf); + d->src = l3->ipv6; + d->dst = ip6->src_addr; + d->len = rte_pktmbuf_pkt_len(mbuf); + d->iface = in_iface; + + edge = ICMP_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static struct rte_node_register frag_needed_node = { + .name = "mpls_push_frag_needed", + .process = mpls_frag_needed_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, +}; + +static struct rte_node_register pkt_too_big_node = { + .name = "mpls_push_pkt_too_big", + .process = mpls_pkt_too_big_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp6_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, +}; + +static struct gr_node_info frag_needed_info = { + .node = &frag_needed_node, + .type = GR_NODE_T_L3, +}; + +static struct gr_node_info pkt_too_big_info = { + .node = &pkt_too_big_node, + .type = GR_NODE_T_L3, +}; + +GR_NODE_REGISTER(frag_needed_info); +GR_NODE_REGISTER(pkt_too_big_info); + +static struct rte_node_register output_frag_needed_node = { + .name = "mpls_output_frag_needed", + .process = mpls_output_frag_needed_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, +}; + +static struct rte_node_register output_pkt_too_big_node = { + .name = "mpls_output_pkt_too_big", + .process = mpls_output_pkt_too_big_process, + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ICMP_OUTPUT] = "icmp6_output", + [NO_HEADROOM] = "error_no_headroom", + [NO_IP] = "error_no_local_ip", + }, +}; + +static struct gr_node_info output_frag_needed_info = { + .node = &output_frag_needed_node, + .type = GR_NODE_T_L3, +}; + +static struct gr_node_info output_pkt_too_big_info = { + .node = &output_pkt_too_big_node, + .type = GR_NODE_T_L3, +}; + +GR_NODE_REGISTER(output_frag_needed_info); +GR_NODE_REGISTER(output_pkt_too_big_info); diff --git a/modules/mpls/datapath/mpls_input.c b/modules/mpls/datapath/mpls_input.c new file mode 100644 index 000000000..c4fd1aa18 --- /dev/null +++ b/modules/mpls/datapath/mpls_input.c @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "checksum.h" +#include "eth.h" +#include "graph.h" +#include "l3.h" +#include "mbuf.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include +#include +#include + +enum { + MPLS_OUTPUT = 0, + IP_INPUT, + IP6_INPUT, + TTL_EXCEEDED, + NO_ROUTE, + BAD_LABEL, + NO_HEADROOM, + EDGE_COUNT, +}; + +struct trace_mpls_data { + uint32_t label; + uint8_t tc; + uint8_t bs; + uint8_t ttl; +}; + +static int mpls_trace_format(char *buf, size_t len, const void *data, size_t /*data_len*/) { + const struct trace_mpls_data *t = data; + return snprintf(buf, len, "label=%u tc=%u bs=%u ttl=%u", t->label, t->tc, t->bs, t->ttl); +} + +static uint16_t +mpls_input_process(struct rte_graph *graph, struct rte_node *node, void **objs, uint16_t nb_objs) { + const struct nexthop_info_mpls *info; + struct nexthop_info_group *nhg; + struct rte_mpls_hdr trace_hdr; + const struct iface *iface; + struct rte_mpls_hdr *mpls; + const struct nexthop *nh; + struct rte_ipv6_hdr *ip6; + struct rte_ipv4_hdr *ip; + struct rte_mbuf *mbuf; + addr_family_t af; + rte_edge_t edge; + uint32_t label; + uint8_t ver2; + uint8_t bos; + uint8_t ver; + uint8_t ttl; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + edge = BAD_LABEL; + + trace_hdr = *rte_pktmbuf_mtod(mbuf, struct rte_mpls_hdr *); + + for (uint8_t depth = 0; depth < GR_MPLS_MAX_STACK_DEPTH; depth++) { + mpls = rte_pktmbuf_mtod(mbuf, struct rte_mpls_hdr *); + label = mpls_hdr_get_label(mpls); + ttl = mpls->ttl; + bos = mpls->bs; + + if (ttl <= 1) { + edge = TTL_EXCEEDED; + break; + } + ttl -= 1; + + if (label < GR_MPLS_LABEL_FIRST_UNRESERVED) { + rte_pktmbuf_adj(mbuf, sizeof(*mpls)); + switch (label) { + case GR_MPLS_LABEL_IPV4_EXPLICIT_NULL: + mbuf->packet_type = RTE_PTYPE_L3_IPV4; + edge = IP_INPUT; + break; + case GR_MPLS_LABEL_IPV6_EXPLICIT_NULL: + mbuf->packet_type = RTE_PTYPE_L3_IPV6; + edge = IP6_INPUT; + break; + case GR_MPLS_LABEL_IMPLICIT_NULL:; + ver = *rte_pktmbuf_mtod(mbuf, uint8_t *) >> 4; + if (ver == 4) { + mbuf->packet_type = RTE_PTYPE_L3_IPV4; + edge = IP_INPUT; + } else if (ver == 6) { + mbuf->packet_type = RTE_PTYPE_L3_IPV6; + edge = IP6_INPUT; + } else { + edge = BAD_LABEL; + } + break; + case GR_MPLS_LABEL_ROUTER_ALERT: + // RFC 2711: strip label and process payload if BOS, + // otherwise continue to next label in stack. + if (bos) { + ver2 = *rte_pktmbuf_mtod(mbuf, uint8_t *) >> 4; + if (ver2 == 4) { + mbuf->packet_type = RTE_PTYPE_L3_IPV4; + edge = IP_INPUT; + } else if (ver2 == 6) { + mbuf->packet_type = RTE_PTYPE_L3_IPV6; + edge = IP6_INPUT; + } else { + edge = BAD_LABEL; + } + break; + } + continue; + default: + edge = BAD_LABEL; + break; + } + break; + } + + iface = mbuf_data(mbuf)->iface; + nh = mpls_fib_lookup(iface->vrf_id, label); + if (nh == NULL) { + edge = NO_ROUTE; + break; + } + + if (nh->type == GR_NH_T_GROUP) { + nhg = nexthop_info_group(nh); + nh = nexthop_group_get_nh(nhg, mbuf->hash.rss); + if (nh == NULL) { + edge = NO_ROUTE; + break; + } + } + + info = nexthop_info_mpls(nh); + + if (info->n_labels > 0) { + if (info->n_labels > 1) { + mpls = gr_mbuf_prepend( + mbuf, mpls, (info->n_labels - 1) * sizeof(*mpls) + ); + if (unlikely(mpls == NULL)) { + edge = NO_HEADROOM; + break; + } + } + for (uint8_t k = 0; k < info->n_labels; k++) { + mpls_hdr_set_label(&mpls[k], info->labels[k]); + mpls[k].ttl = ttl; + if (k < info->n_labels - 1) { + mpls[k].bs = 0; + mpls[k].tc = 0; + } + } + l3_mbuf_data(mbuf)->nh = nh; + mbuf->packet_type = RTE_PTYPE_TUNNEL_MPLS_IN_GRE; + edge = MPLS_OUTPUT; + break; + } + + if (bos) { + if (info->via_nh == NULL) { + edge = NO_ROUTE; + break; + } + rte_pktmbuf_adj(mbuf, sizeof(*mpls)); + af = info->payload_af; + if (af == GR_AF_UNSPEC) { + ver = *rte_pktmbuf_mtod(mbuf, uint8_t *) >> 4; + if (ver == 4) + af = GR_AF_IP4; + else if (ver == 6) + af = GR_AF_IP6; + } + if (af == GR_AF_IP4) { + ip = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + ip->hdr_checksum = fixup_checksum_16( + ip->hdr_checksum, + rte_cpu_to_be_16(ip->time_to_live << 8), + rte_cpu_to_be_16(ttl << 8) + ); + ip->time_to_live = ttl; + mbuf->packet_type = RTE_PTYPE_L3_IPV4; + } else if (af == GR_AF_IP6) { + ip6 = rte_pktmbuf_mtod(mbuf, struct rte_ipv6_hdr *); + ip6->hop_limits = ttl; + mbuf->packet_type = RTE_PTYPE_L3_IPV6; + } else { + edge = BAD_LABEL; + break; + } + l3_mbuf_data(mbuf)->nh = info->via_nh; + edge = MPLS_OUTPUT; + break; + } + + rte_pktmbuf_adj(mbuf, sizeof(*mpls)); + } + + if (gr_mbuf_is_traced(mbuf)) { + struct trace_mpls_data *t = gr_mbuf_trace_add(mbuf, node, sizeof(*t)); + t->label = mpls_hdr_get_label(&trace_hdr); + t->tc = trace_hdr.tc; + t->bs = trace_hdr.bs; + t->ttl = trace_hdr.ttl; + } + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static void mpls_input_register(void) { + gr_eth_input_add_type(RTE_BE16(RTE_ETHER_TYPE_MPLS), "mpls_input"); +} + +static struct rte_node_register mpls_input_node = { + .name = "mpls_input", + + .process = mpls_input_process, + + .nb_edges = EDGE_COUNT, + .next_nodes = { + [MPLS_OUTPUT] = "mpls_output", + [IP_INPUT] = "ip_input", + [IP6_INPUT] = "ip6_input", + [TTL_EXCEEDED] = "mpls_input_ttl_exceeded", + [NO_ROUTE] = "mpls_input_no_route", + [BAD_LABEL] = "mpls_input_bad_label", + [NO_HEADROOM] = "error_no_headroom", + }, +}; + +static struct gr_node_info info = { + .node = &mpls_input_node, + .type = GR_NODE_T_L3, + .trace_format = mpls_trace_format, + .register_callback = mpls_input_register, +}; + +GR_NODE_REGISTER(info); + +GR_DROP_REGISTER(mpls_input_ttl_exceeded); +GR_DROP_REGISTER(mpls_input_no_route); +GR_DROP_REGISTER(mpls_input_bad_label); diff --git a/modules/mpls/datapath/mpls_output.c b/modules/mpls/datapath/mpls_output.c new file mode 100644 index 000000000..b5bb3dfd9 --- /dev/null +++ b/modules/mpls/datapath/mpls_output.c @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "control_input.h" +#include "eth.h" +#include "graph.h" +#include "l3.h" +#include "log.h" +#include "mbuf.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include +#include +#include +#include +#include + +LOG_TYPE("mpls"); + +static rte_edge_t mpls_output_ctrl; + +int mpls_resubmit_cb(struct rte_mbuf *m, struct nexthop *) { + l3_mbuf_data(m)->nh = mpls_hold_mbuf_data(m)->mpls_nh; + mbuf_data(m)->iface = NULL; + if (post_to_stack(mpls_output_ctrl, m) < 0) { + LOG(ERR, "post_to_stack: %s", strerror(errno)); + return -errno; + } + return 0; +} + +enum { + ETH_OUTPUT = 0, + HOLD, + NO_ROUTE, + TRANSIT_FRAG_NEEDED, + TRANSIT_PKT_TOO_BIG, + MTU_EXCEEDED, + EDGE_COUNT, +}; + +static uint16_t +mpls_output_process(struct rte_graph *graph, struct rte_node *node, void **objs, uint16_t nb_objs) { + struct eth_output_mbuf_data *eth_data; + const struct nexthop_info_l3 *l3; + const struct rte_ipv4_hdr *ip4; + const struct nexthop *via_nh; + const struct iface *iface; + const struct nexthop *nh; + struct rte_mpls_hdr *m; + struct rte_mbuf *mbuf; + rte_edge_t edge; + uint32_t depth; + uint8_t ver; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + nh = l3_mbuf_data(mbuf)->nh; + if (nh == NULL) { + edge = NO_ROUTE; + goto next; + } + + if (nh->type == GR_NH_T_MPLS) { + via_nh = nexthop_info_mpls(nh)->via_nh; + } else if (nh->type == GR_NH_T_L3) { + via_nh = nh; + } else { + edge = NO_ROUTE; + goto next; + } + + if (via_nh == NULL) { + edge = NO_ROUTE; + goto next; + } + + l3 = nexthop_info_l3(via_nh); + iface = iface_from_id(via_nh->iface_id); + if (iface == NULL) { + edge = NO_ROUTE; + goto next; + } + + if (rte_pktmbuf_pkt_len(mbuf) > iface->mtu) { + m = rte_pktmbuf_mtod(mbuf, struct rte_mpls_hdr *); + depth = 0; + while (depth < GR_MPLS_MAX_STACK_DEPTH && !m->bs) { + m++; + depth++; + } + ver = ((const uint8_t *)(m + 1))[0] >> 4; + if (ver == 6) { + edge = TRANSIT_PKT_TOO_BIG; + } else if (ver == 4) { + ip4 = (const struct rte_ipv4_hdr *)(m + 1); + edge = (ip4->fragment_offset + & rte_cpu_to_be_16(RTE_IPV4_HDR_DF_FLAG)) ? + TRANSIT_FRAG_NEEDED : + MTU_EXCEEDED; + } else { + edge = MTU_EXCEEDED; + } + goto next; + } + + if (l3->state != GR_NH_S_REACHABLE) { + mpls_hold_mbuf_data(mbuf)->mpls_nh = nh; + l3_mbuf_data(mbuf)->nh = via_nh; + mbuf->packet_type |= RTE_PTYPE_TUNNEL_MPLS_IN_GRE; + edge = HOLD; + goto next; + } + + eth_data = eth_output_mbuf_data(mbuf); + eth_data->dst = l3->mac; + if (mbuf->packet_type & RTE_PTYPE_L3_IPV4) + eth_data->ether_type = RTE_BE16(RTE_ETHER_TYPE_IPV4); + else if (mbuf->packet_type & RTE_PTYPE_L3_IPV6) + eth_data->ether_type = RTE_BE16(RTE_ETHER_TYPE_IPV6); + else + eth_data->ether_type = RTE_BE16(RTE_ETHER_TYPE_MPLS); + mbuf_data(mbuf)->iface = iface; + edge = ETH_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static struct rte_node_register mpls_output_node_reg = { + .name = "mpls_output", + + .process = mpls_output_process, + + .nb_edges = EDGE_COUNT, + .next_nodes = { + [ETH_OUTPUT] = "eth_output", + [HOLD] = "ip_hold", + [NO_ROUTE] = "mpls_output_no_route", + [TRANSIT_FRAG_NEEDED] = "mpls_output_frag_needed", + [TRANSIT_PKT_TOO_BIG] = "mpls_output_pkt_too_big", + [MTU_EXCEEDED] = "mpls_output_mtu_exceeded", + }, +}; + +static void mpls_output_register(void) { + mpls_output_ctrl = gr_control_input_register_handler("mpls_output"); +} + +static struct gr_node_info info = { + .node = &mpls_output_node_reg, + .type = GR_NODE_T_L3, + .register_callback = mpls_output_register, +}; + +GR_NODE_REGISTER(info); + +GR_DROP_REGISTER(mpls_output_no_route); +GR_DROP_REGISTER(mpls_output_mtu_exceeded); diff --git a/modules/mpls/datapath/mpls_push.c b/modules/mpls/datapath/mpls_push.c new file mode 100644 index 000000000..5749bb177 --- /dev/null +++ b/modules/mpls/datapath/mpls_push.c @@ -0,0 +1,137 @@ +// SPDX-License-Identifier: BSD-3-Clause +// Copyright (c) 2025 Matej Muzila + +#include "graph.h" +#include "ip4_datapath.h" +#include "ip6_datapath.h" +#include "l3.h" +#include "mbuf.h" +#include "mpls.h" +#include "mpls_datapath.h" + +#include + +#include +#include + +enum { + MPLS_OUTPUT = 0, + NO_HEADROOM, + FRAG_NEEDED, + PKT_TOO_BIG, + MTU_EXCEEDED, + EDGE_COUNT, +}; + +static uint16_t +mpls_push_process(struct rte_graph *graph, struct rte_node *node, void **objs, uint16_t nb_objs) { + const struct nexthop_info_mpls *info; + struct rte_mpls_hdr *mpls; + const struct iface *iface; + struct rte_ipv6_hdr *ip6; + struct rte_ipv4_hdr *ip4; + const struct nexthop *nh; + struct rte_mbuf *mbuf; + uint16_t overhead; + rte_edge_t edge; + uint8_t ttl; + uint8_t tc; + + for (uint16_t i = 0; i < nb_objs; i++) { + mbuf = objs[i]; + + nh = l3_mbuf_data(mbuf)->nh; + info = nexthop_info_mpls(nh); + + if (info->n_labels == 0 || info->via_nh == NULL) { + edge = NO_HEADROOM; + goto next; + } + + iface = iface_from_id(info->via_nh->iface_id); + if (iface == NULL) { + edge = NO_HEADROOM; + goto next; + } + + overhead = info->n_labels * sizeof(struct rte_mpls_hdr); + if (rte_pktmbuf_pkt_len(mbuf) + overhead > iface->mtu) { + if (mbuf->packet_type & RTE_PTYPE_L3_IPV6) { + edge = PKT_TOO_BIG; + } else { + ip4 = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + edge = (ip4->fragment_offset + & rte_cpu_to_be_16(RTE_IPV4_HDR_DF_FLAG)) ? + FRAG_NEEDED : + MTU_EXCEEDED; + } + goto next; + } + + if (mbuf->packet_type & RTE_PTYPE_L3_IPV4) { + ip4 = rte_pktmbuf_mtod(mbuf, struct rte_ipv4_hdr *); + ttl = ip4->time_to_live; + tc = ip4->type_of_service >> 5; + } else { + ip6 = rte_pktmbuf_mtod(mbuf, struct rte_ipv6_hdr *); + ttl = ip6->hop_limits; + tc = (rte_be_to_cpu_32(ip6->vtc_flow) >> 25) & 0x7; + } + + if (info->ttl) + ttl = info->ttl; + + mpls = gr_mbuf_prepend(mbuf, mpls, (info->n_labels - 1) * sizeof(*mpls)); + if (unlikely(mpls == NULL)) { + edge = NO_HEADROOM; + goto next; + } + + for (uint8_t j = 0; j < info->n_labels; j++) { + mpls_hdr_set_label(&mpls[j], info->labels[j]); + mpls[j].tc = tc; + mpls[j].bs = (j == info->n_labels - 1) ? 1 : 0; + mpls[j].ttl = ttl; + } + + l3_mbuf_data(mbuf)->nh = nh; + mbuf->packet_type = RTE_PTYPE_TUNNEL_MPLS_IN_GRE; + edge = MPLS_OUTPUT; +next: + if (gr_mbuf_is_traced(mbuf)) + gr_mbuf_trace_add(mbuf, node, 0); + rte_node_enqueue_x1(graph, node, edge, mbuf); + } + + return nb_objs; +} + +static void mpls_push_register(void) { + ip_output_register_nexthop_type(GR_NH_T_MPLS, "mpls_push"); + ip6_output_register_nexthop_type(GR_NH_T_MPLS, "mpls_push"); +} + +static struct rte_node_register mpls_push_node = { + .name = "mpls_push", + + .process = mpls_push_process, + + .nb_edges = EDGE_COUNT, + .next_nodes = { + [MPLS_OUTPUT] = "mpls_output", + [NO_HEADROOM] = "error_no_headroom", + [FRAG_NEEDED] = "mpls_push_frag_needed", + [PKT_TOO_BIG] = "mpls_push_pkt_too_big", + [MTU_EXCEEDED] = "mpls_push_mtu_exceeded", + }, +}; + +static struct gr_node_info info = { + .node = &mpls_push_node, + .type = GR_NODE_T_L3, + .register_callback = mpls_push_register, +}; + +GR_NODE_REGISTER(info); + +GR_DROP_REGISTER(mpls_push_mtu_exceeded); diff --git a/modules/mpls/meson.build b/modules/mpls/meson.build new file mode 100644 index 000000000..8bc77a9c3 --- /dev/null +++ b/modules/mpls/meson.build @@ -0,0 +1,7 @@ +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +subdir('api') +subdir('cli') +subdir('control') +subdir('datapath') diff --git a/modules/policy/datapath/nat_datapath.h b/modules/policy/datapath/nat_datapath.h index 1e791d46d..8d0319740 100644 --- a/modules/policy/datapath/nat_datapath.h +++ b/modules/policy/datapath/nat_datapath.h @@ -3,6 +3,7 @@ #pragma once +#include "checksum.h" #include "iface.h" #include "nexthop.h" @@ -16,35 +17,6 @@ GR_NH_TYPE_INFO(GR_NH_T_DNAT, nexthop_info_dnat, { struct nexthop *arp; }); -static inline rte_be16_t -fixup_checksum_16(rte_be16_t old_cksum, rte_be16_t old_field, rte_be16_t new_field) { - uint32_t sum; - - // RFC 1624: HC' = ~(~HC + ~m + m') - // Note: 1's complement sum is endian-independent (RFC 1071, page 2). - sum = ~old_cksum & 0xffff; - sum += (~old_field & 0xffff) + new_field; - sum = (sum >> 16) + (sum & 0xffff); - sum += (sum >> 16); - - return ~sum & 0xffff; -} - -static inline rte_be16_t -fixup_checksum_32(rte_be16_t old_cksum, ip4_addr_t old_addr, ip4_addr_t new_addr) { - uint32_t sum; - - // Checksum 32-bit datum as as two 16-bit. Note, the first - // 32->16 bit reduction is not necessary. - sum = ~old_cksum & 0xffff; - sum += (~old_addr & 0xffff) + (new_addr & 0xffff); - sum += (~old_addr >> 16) + (new_addr >> 16); - sum = (sum >> 16) + (sum & 0xffff); - sum += (sum >> 16); - - return ~sum & 0xffff; -} - typedef enum { NAT_VERDICT_CONTINUE, NAT_VERDICT_FINAL, diff --git a/smoke/isis_sr_frr_test.sh b/smoke/isis_sr_frr_test.sh new file mode 100755 index 000000000..99e1484f5 --- /dev/null +++ b/smoke/isis_sr_frr_test.sh @@ -0,0 +1,121 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +# Verify that IS-IS Segment Routing automatically installs LSPs in grout via +# dplane_grout. isisd computes the penultimate-hop LSP for the peer's node SID +# and sends it to zebra → dplane_grout → grout LFIB. +# +# Both grout and the peer use SRGB [16000, 23999]: +# grout prefix SID index 0 → label 16000 (explicit-null toward grout) +# peer prefix SID index 1 → label 16001 (PHP at grout for peer) +# +# IS-IS SR installs at grout: in=16001, PHP (implicit-null), via 172.16.0.2 +# +# .--------------------. +# | netns "isis-peer" | +# .--------..------------. | .-------. | +# | zebra || grout | | | isisd | | +# '--------'| | | '-------' | +# .-------. | .------------. .------------. .-------. | +# | isisd | | | p0 | net_tap | x-p0 | | zebra | | +# '-------' | | +-------------+ | '-------' | +# .------. | 172.16.0.1 | | 172.16.0.2 |.----------. | +# | main | '------------' '------------'| lo | | +# '------' | | | | | +# | ping <------------------------------------> | 16.0.0.1 | | +# | | | '----------' | +# '-----------' '--------------------' + +. $(dirname $0)/_init_frr.sh + +create_interface p0 +set_ip_address p0 172.16.0.1/24 + +# Configure grout's FRR: IS-IS with Segment Routing +vtysh <<-EOF +configure terminal +! +ip router-id 172.16.0.1 +! +interface lo + ip address 17.0.0.1/32 +exit +! +interface p0 + ip router isis smoke + isis network point-to-point +exit +! +router isis smoke + net 49.0000.0000.0001.00 + is-type level-2-only + redistribute ipv4 connected level-2 + segment-routing on + segment-routing global-block 16000 23999 + segment-routing node-msd 8 + segment-routing prefix 17.0.0.1/32 index 0 explicit-null +exit +! +EOF + +start_frr isis-peer 0 +ip link set x-p0 netns isis-peer + +# Enable kernel MPLS in the peer netns so that FRR zebra can install the +# pop rule for its own node SID (label 16001) when IS-IS SR comes up. +ip netns exec isis-peer sysctl -wq net.mpls.platform_labels=1000 +ip netns exec isis-peer sysctl -wq net.mpls.conf.x-p0.input=1 + +vtysh -N isis-peer <<-EOF +configure terminal +! +ip router-id 172.16.0.2 +! +interface lo + ip address 16.0.0.1/32 +exit +! +interface x-p0 + ip address 172.16.0.2/24 + ip router isis smoke + isis network point-to-point +exit +! +router isis smoke + net 49.0000.0000.0002.00 + is-type level-2-only + redistribute ipv4 connected level-2 + segment-routing on + segment-routing global-block 16000 23999 + segment-routing node-msd 8 + segment-routing prefix 16.0.0.1/32 index 1 +exit +! +EOF + +# Wait for IS-IS adjacency +attempts=60 +while ! vtysh -c 'show isis neighbor json' | jq -e '.areas[0].circuits[0].state == "Up"'; do + sleep 1 + if [ "$attempts" -le 0 ]; then + fail "IS-IS failed to connect to neighbor." + fi + attempts=$((attempts - 1)) +done + +# Wait for IS-IS route exchange +attempts=90 +while ! vtysh -c 'show ip route isis json' | jq -e '."16.0.0.1/32"'; do + sleep 1 + if [ "$attempts" -le 0 ]; then + fail "IS-IS failed to get routes." + fi + attempts=$((attempts - 1)) +done + +# IS-IS SR installs LSP for peer's node SID (label 16001) in grout LFIB +wait_event -t 60 'mpls route add: vrf=main label=16001 .* origin=isis' + +# IP connectivity via IS-IS learned route +grcli ping 16.0.0.1 count 3 delay 10 diff --git a/smoke/mpls_cli_test.sh b/smoke/mpls_cli_test.sh new file mode 100755 index 000000000..8b9173ab7 --- /dev/null +++ b/smoke/mpls_cli_test.sh @@ -0,0 +1,56 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +grcli address add 172.16.0.1/24 iface p0 + +netns_add n0 +move_to_netns x-p0 n0 +ip -n n0 addr add 172.16.0.2/24 dev x-p0 + +# create MPLS nexthop with single label +grcli nexthop add mpls iface p0 via 172.16.0.2 labels 100 id 10 + +# verify nexthop is visible +grcli nexthop show id 10 + +# create MPLS nexthop with multiple labels +grcli nexthop add mpls iface p0 via 172.16.0.2 labels 100 200 300 id 20 + +# verify multi-label nexthop +grcli nexthop show id 20 + +# create pop nexthop (no labels) +grcli nexthop add mpls iface p0 via 172.16.0.2 id 30 + +# verify pop nexthop +grcli nexthop show id 30 + +# list MPLS nexthops +grcli nexthop show type mpls + +# add label route +grcli mpls route add 500 nexthop id 10 + +# get label route +grcli mpls route get 500 + +# list label routes +grcli mpls route show + +# idempotent add (exist_ok) +grcli mpls route add 500 nexthop id 10 + +# delete label route +grcli mpls route del 500 + +# idempotent delete (missing_ok) +grcli mpls route del 500 + +# cleanup nexthops +grcli nexthop del 10 +grcli nexthop del 20 +grcli nexthop del 30 diff --git a/smoke/mpls_explicit_null6_test.sh b/smoke/mpls_explicit_null6_test.sh new file mode 100755 index 000000000..2b83ab3f4 --- /dev/null +++ b/smoke/mpls_explicit_null6_test.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 2001:db8:0::1/64 iface p0 +grcli address add 2001:db8:1::1/64 iface p1 + +# IPv6 route for the decapsulated packet (no MPLS nexthop needed) +grcli route add fd00::/64 via 2001:db8:1::2 + +# return path +grcli route add fd00:0:0:1::/64 via 2001:db8:0::2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 2001:db8:$n::2/64 dev $p +done + +# n0: push IPv6 Explicit NULL (label 2) +ip -n n0 addr add fd00:0:0:1::1/128 dev lo +ip -n n0 route add fd00::/64 encap mpls 2 via 2001:db8:0::1 dev x-p0 + +# n1: plain IPv6 receiver +ip -n n1 addr add fd00::1/128 dev lo +ip -n n1 route add default via 2001:db8:1::1 + +# resolve NDP before MPLS test +ip netns exec n0 ping6 -c1 -n 2001:db8:0::1 + +# n0 pushes MPLS(label=2) -> grout strips and routes via ip6_input -> n1 receives plain IPv6 +ip netns exec n0 ping6 -i0.01 -c3 -n fd00::1 diff --git a/smoke/mpls_explicit_null_test.sh b/smoke/mpls_explicit_null_test.sh new file mode 100755 index 000000000..c7ae24825 --- /dev/null +++ b/smoke/mpls_explicit_null_test.sh @@ -0,0 +1,38 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# IP route for the decapsulated packet (no MPLS nexthop needed) +grcli route add 10.0.1.0/24 via 172.16.1.2 + +# return path +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push explicit null (label 0) +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 0 via 172.16.0.1 dev x-p0 + +# n1: plain IP receiver +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# n0 pushes MPLS(label=0) -> grout strips and routes via ip_input -> n1 receives plain IP +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_frag_needed_test.sh b/smoke/mpls_frag_needed_test.sh new file mode 100755 index 000000000..7e910cbef --- /dev/null +++ b/smoke/mpls_frag_needed_test.sh @@ -0,0 +1,53 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: push label 100 (4 bytes overhead), forward via n1 +# Effective MTU = iface_mtu(1500) - 1*4 = 1496 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 100 id 42 + +# IP route pointing to MPLS nexthop +grcli route add 10.0.1.0/24 via id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: plain IP sender +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add default via 172.16.0.1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 100 dev lo +ip -n n1 route add default via 172.16.1.1 + +# verify basic connectivity first +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 + +# send oversized IPv4 packet with DF set: payload=1477 -> IP=1497 -> with label=1501 > 1500 +# ping -s sets payload size; total IP = payload + 28 (ICMP+IP headers) +# we need IP size > 1496 (effective MTU), so payload > 1468 +# use payload=1470 -> IP=1498 > 1496 +output=$(ip netns exec n0 ping -c1 -W2 -s1470 -M do -n 10.0.1.1 2>&1 || true) +echo "$output" +echo "$output" | grep -qi "frag needed\|message too big\|mtu\|unreachable" \ + || fail "expected ICMP Frag Needed for oversized DF packet, got: $output" + +echo "ICMP Frag Needed correctly received for oversized DF packet" diff --git a/smoke/mpls_hold_test.sh b/smoke/mpls_hold_test.sh new file mode 100755 index 000000000..bc3275142 --- /dev/null +++ b/smoke/mpls_hold_test.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: push label 100, forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 100 id 42 + +# IP route pointing to MPLS nexthop +grcli route add 10.0.1.0/24 via id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: plain IP sender +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add default via 172.16.0.1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 100 dev lo +ip -n n1 route add default via 172.16.1.1 + +# no ARP pre-warm: first packets will hit the hold path while ARP resolves +# use longer timeout (-W5) and several packets so ARP resolves during the burst +ip netns exec n0 ping -i0.1 -c10 -W5 -n 10.0.1.1 diff --git a/smoke/mpls_multi_label_push_test.sh b/smoke/mpls_multi_label_push_test.sh new file mode 100755 index 000000000..c612fd67b --- /dev/null +++ b/smoke/mpls_multi_label_push_test.sh @@ -0,0 +1,46 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: push two labels (outer=200 inner=100), forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 200 100 id 42 + +# IP route pointing to multi-label MPLS nexthop +grcli route add 10.0.1.0/24 via id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: plain IP sender +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add default via 172.16.0.1 + +# n1: MPLS receiver, pop outer label 200 then inner label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip netns exec n1 sysctl -wq net.mpls.conf.lo.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +# pop outer 200 via lo (recirculates as [100,IP] to lo), then pop inner 100 +ip -n n1 -f mpls route add 200 dev lo +ip -n n1 -f mpls route add 100 dev lo + +# return path from n1 +ip -n n1 route add default via 172.16.1.1 + +# plain IP -> grout pushes MPLS(200,100) -> n1 pops both labels +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_pkt_too_big_test.sh b/smoke/mpls_pkt_too_big_test.sh new file mode 100755 index 000000000..c63be29e8 --- /dev/null +++ b/smoke/mpls_pkt_too_big_test.sh @@ -0,0 +1,53 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 2001:db8:0::1/64 iface p0 +grcli address add 2001:db8:1::1/64 iface p1 + +# MPLS nexthop: push label 100 (4 bytes overhead), forward via n1 +# Effective MTU = iface_mtu(1500) - 1*4 = 1496 +grcli nexthop add mpls iface p1 via 2001:db8:1::2 labels 100 id 42 + +# IPv6 route pointing to MPLS nexthop +grcli route add fd00::/64 via id 42 + +# return path (plain IPv6) +grcli route add fd00:0:0:1::/64 via 2001:db8:0::2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 2001:db8:$n::2/64 dev $p +done + +# n0: plain IPv6 sender +ip -n n0 addr add fd00:0:0:1::1/128 dev lo +ip -n n0 route add default via 2001:db8:0::1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add fd00::1/128 dev lo +ip -n n1 -f mpls route add 100 dev lo +ip -n n1 route add default via 2001:db8:1::1 + +# verify basic connectivity first +ip netns exec n0 ping6 -i0.01 -c3 -n fd00::1 + +# send oversized IPv6 packet: payload=1450 -> IP=1490+40=1530 > 1496 effective MTU +# IPv6 always has DF semantics; ICMPv6 Packet Too Big must be returned +# ping6 -s sets data size; total IPv6 = data + 8 (ICMPv6) + 40 (IPv6 header) +# need IPv6 size > 1496, so data > 1448; use data=1450 -> IPv6=1498 > 1496 +output=$(ip netns exec n0 ping6 -c1 -W2 -s1450 -n fd00::1 2>&1 || true) +echo "$output" +echo "$output" | grep -qi "too big\|packet too big\|mtu\|unreachable" \ + || fail "expected ICMPv6 Packet Too Big for oversized IPv6 packet, got: $output" + +echo "ICMPv6 Packet Too Big correctly received for oversized IPv6 packet" diff --git a/smoke/mpls_pop6_test.sh b/smoke/mpls_pop6_test.sh new file mode 100755 index 000000000..6525ac03f --- /dev/null +++ b/smoke/mpls_pop6_test.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 2001:db8:0::1/64 iface p0 +grcli address add 2001:db8:1::1/64 iface p1 + +# MPLS nexthop: pop (no labels = PHP/pop), forward via n1 +grcli nexthop add mpls iface p1 via 2001:db8:1::2 id 42 + +# label route: incoming label 100 -> pop +grcli mpls route add 100 nexthop id 42 + +# return path (plain IPv6) +grcli route add fd00:0:0:1::/64 via 2001:db8:0::2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 2001:db8:$n::2/64 dev $p +done + +# n0: push label 100 for IPv6 traffic +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add fd00:0:0:1::1/128 dev lo +ip -n n0 route add fd00::/64 encap mpls 100 via 2001:db8:0::1 dev x-p0 + +# n1: plain IPv6 receiver +ip -n n1 addr add fd00::1/128 dev lo +ip -n n1 route add default via 2001:db8:1::1 + +# resolve NDP before MPLS test +ip netns exec n0 ping6 -c1 -n 2001:db8:0::1 + +# n0 pushes MPLS(100) -> grout pops -> n1 receives plain IPv6 +ip netns exec n0 ping6 -i0.01 -c3 -n fd00::1 diff --git a/smoke/mpls_pop_payload_test.sh b/smoke/mpls_pop_payload_test.sh new file mode 100755 index 000000000..64c892c88 --- /dev/null +++ b/smoke/mpls_pop_payload_test.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: pop with explicit payload type +grcli nexthop add mpls iface p1 via 172.16.1.2 id 42 payload ipv4 + +# label route: incoming label 100 -> pop +grcli mpls route add 100 nexthop id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push label 100 when sending to 10.0.1.0/24 +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 dev x-p0 + +# n1: plain IP receiver +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# n0 pushes MPLS(100) -> grout pops using explicit payload type -> n1 receives plain IP +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_pop_test.sh b/smoke/mpls_pop_test.sh new file mode 100755 index 000000000..89b8df967 --- /dev/null +++ b/smoke/mpls_pop_test.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: pop (no labels), forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 id 42 + +# label route: incoming label 100 -> pop +grcli mpls route add 100 nexthop id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push label 100 when sending to 10.0.1.0/24 +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 dev x-p0 + +# n1: plain IP receiver +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# n0 pushes MPLS(100) -> grout pops label, propagates TTL -> n1 receives plain IP +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_push6_test.sh b/smoke/mpls_push6_test.sh new file mode 100755 index 000000000..82699868a --- /dev/null +++ b/smoke/mpls_push6_test.sh @@ -0,0 +1,43 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 2001:db8:0::1/64 iface p0 +grcli address add 2001:db8:1::1/64 iface p1 + +# MPLS nexthop: push label 100 for IPv6 traffic, forward via n1 +grcli nexthop add mpls iface p1 via 2001:db8:1::2 labels 100 id 42 + +# IPv6 route pointing to MPLS nexthop +grcli route add fd00::/64 via id 42 + +# return path (plain IPv6) +grcli route add fd00:0:0:1::/64 via 2001:db8:0::2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 2001:db8:$n::2/64 dev $p +done + +# n0: plain IPv6 sender +ip -n n0 addr add fd00:0:0:1::1/128 dev lo +ip -n n0 route add default via 2001:db8:0::1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add fd00::1/128 dev lo +ip -n n1 -f mpls route add 100 dev lo + +# return path from n1 +ip -n n1 route add default via 2001:db8:1::1 + +# plain IPv6 -> grout pushes MPLS label -> n1 pops label +ip netns exec n0 ping6 -i0.01 -c3 -n fd00::1 diff --git a/smoke/mpls_push_test.sh b/smoke/mpls_push_test.sh new file mode 100755 index 000000000..69e37bf9f --- /dev/null +++ b/smoke/mpls_push_test.sh @@ -0,0 +1,43 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: push label 100, forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 100 id 42 + +# IP route pointing to MPLS nexthop +grcli route add 10.0.1.0/24 via id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: plain IP sender +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add default via 172.16.0.1 + +# n1: MPLS receiver, pop label 100 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 100 dev lo + +# return path from n1 +ip -n n1 route add default via 172.16.1.1 + +# plain IP -> grout pushes MPLS label -> n1 pops label +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_static_frr_test.sh b/smoke/mpls_static_frr_test.sh new file mode 100755 index 000000000..2260f0374 --- /dev/null +++ b/smoke/mpls_static_frr_test.sh @@ -0,0 +1,83 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +# Verify the FRR staticd → dplane_grout → grout MPLS LSP path for two cases: +# 1. PHP (implicit-null): in-label 100, grout pops and forwards to n1 +# 2. Label swap: in-label 100 → out-label 200, grout swaps and n1 pops +# +# .-------..--------------. +# | zebra || grout | +# '-------'| | +# .-------.|.-----..------.. +# |staticd||| p0 || p1 || +# '-------'|| 172|| 172 || +# ||16.0.1|16.1.1|| +# .----------------. .-----------. +# | n0 | | n1 | +# | 172.16.0.2/24 | |172.16.1.2 | +# | push MPLS 100 | |10.0.1.1/lo| +# '----------------' '-----------' +# +# PHP flow: +# n0 --[label 100]--> grout --[pop]--> n1 (10.0.1.1) +# Swap flow: +# n0 --[label 100]--> grout --[swap→200]--> n1 --[kernel pop 200]--> lo + +. $(dirname $0)/_init_frr.sh + +create_interface p0 +create_interface p1 + +set_ip_address p0 172.16.0.1/24 +set_ip_address p1 172.16.1.1/24 + +netns_add n0 +netns_add n1 +move_to_netns x-p0 n0 +move_to_netns x-p1 n1 + +ip -n n0 addr add 172.16.0.2/24 dev x-p0 +ip -n n1 addr add 172.16.1.2/24 dev x-p1 +ip -n n1 addr add 10.0.1.1/32 dev lo + +# n1: reply path for pings originating from n0 (172.16.0.2) +ip -n n1 route add 172.16.0.0/24 via 172.16.1.1 + +# grout: IP route to 10.0.1.0/24 so that after PHP it can forward to n1 +set_ip_route 10.0.1.0/24 172.16.1.2 + +# Install MPLS LSP via FRR staticd: in-label 100, PHP (implicit-null) via 172.16.1.2 +_apply_frr_config 0 \ + "mpls route add: vrf=main label=100 .* origin=zebra_static" \ + "mpls lsp 100 172.16.1.2 implicit-null" + +# n0: push MPLS label 100 for traffic to 10.0.1.0/24 via grout +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 + +# n0 sends MPLS-labeled packet, grout pops label 100 (PHP) and forwards to n1 +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 + +# Remove LSP and verify removal event +_apply_frr_config 0 \ + "mpls route del: vrf=main label=100 .* origin=zebra_static" \ + "no mpls lsp 100 172.16.1.2 implicit-null" + +# Label swap: in-label 100 → out-label 200 via 172.16.1.2 +# n1 needs kernel MPLS to accept and pop label 200 on ingress +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 -f mpls route add 200 dev lo + +_apply_frr_config 0 \ + "mpls route add: vrf=main label=100 .* origin=zebra_static" \ + "mpls lsp 100 172.16.1.2 200" + +# n0 sends MPLS label 100, grout swaps to 200, n1 pops and delivers to lo +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 + +# Remove swap LSP +_apply_frr_config 0 \ + "mpls route del: vrf=main label=100 .* origin=zebra_static" \ + "no mpls lsp 100 172.16.1.2 200" diff --git a/smoke/mpls_swap_test.sh b/smoke/mpls_swap_test.sh new file mode 100755 index 000000000..9f8df4a25 --- /dev/null +++ b/smoke/mpls_swap_test.sh @@ -0,0 +1,47 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: swap to label 200, forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 200 id 42 + +# label route: incoming label 100 -> swap to 200 +grcli mpls route add 100 nexthop id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push label 100 when sending to 10.0.1.0/24 +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 dev x-p0 + +# n1: pop label 200 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 200 dev lo + +# return path from n1 +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# n0 pushes MPLS(100) -> grout swaps to MPLS(200) -> n1 pops +ip netns exec n0 ping -i0.01 -c3 -n 10.0.1.1 diff --git a/smoke/mpls_ttl_test.sh b/smoke/mpls_ttl_test.sh new file mode 100755 index 000000000..9653e7902 --- /dev/null +++ b/smoke/mpls_ttl_test.sh @@ -0,0 +1,47 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +. $(dirname $0)/_init.sh + +port_add p0 +port_add p1 +grcli address add 172.16.0.1/24 iface p0 +grcli address add 172.16.1.1/24 iface p1 + +# MPLS nexthop: swap to label 200, forward via n1 +grcli nexthop add mpls iface p1 via 172.16.1.2 labels 200 id 42 + +# label route: incoming label 100 -> swap to 200 +grcli mpls route add 100 nexthop id 42 + +# return path (plain IP) +grcli route add 10.0.0.0/24 via 172.16.0.2 + +for n in 0 1; do + p=x-p$n + ns=n$n + netns_add $ns + move_to_netns $p $ns + ip -n $ns addr add 172.16.$n.2/24 dev $p +done + +# n0: push label 100 +ip netns exec n0 sysctl -wq net.mpls.platform_labels=1000 +ip -n n0 addr add 10.0.0.1/32 dev lo +ip -n n0 route add 10.0.1.0/24 encap mpls 100 via 172.16.0.1 dev x-p0 + +# n1: pop label 200 +ip netns exec n1 sysctl -wq net.mpls.platform_labels=1000 +ip netns exec n1 sysctl -wq net.mpls.conf.x-p1.input=1 +ip -n n1 addr add 10.0.1.1/32 dev lo +ip -n n1 -f mpls route add 200 dev lo +ip -n n1 route add default via 172.16.1.1 + +# resolve ARP before MPLS test +ip netns exec n0 ping -c1 -n 172.16.0.1 + +# TTL=1: grout must drop the MPLS-encapped packet and not forward it +ip netns exec n0 ping -c1 -W1 -t1 -n 10.0.1.1 && fail "packet with TTL=1 should not reach destination" + +echo "TTL=1 packet correctly dropped" diff --git a/smoke/ospf_sr_frr_test.sh b/smoke/ospf_sr_frr_test.sh new file mode 100755 index 000000000..dcae4bf9b --- /dev/null +++ b/smoke/ospf_sr_frr_test.sh @@ -0,0 +1,120 @@ +#!/bin/bash +# SPDX-License-Identifier: BSD-3-Clause +# Copyright (c) 2025 Matej Muzila + +# Verify that OSPF Segment Routing automatically installs LSPs in grout via +# dplane_grout. ospfd computes the penultimate-hop LSP for the peer's node SID +# and sends it to zebra → dplane_grout → grout LFIB. +# +# Both grout and the peer use SRGB [16000, 23999]: +# grout prefix SID index 0 → label 16000 (or 15000 depending on ospfd timing) +# peer prefix SID index 1 → label 16001 (or 15001 depending on ospfd timing) +# +# OSPF-SR installs at grout: in=peer's SID, PHP (implicit-null), via 172.16.0.2 +# +# .--------------------. +# | netns "ospf-peer" | +# .--------..------------. | .-------. | +# | zebra || grout | | | ospfd | | +# '--------'| | | '-------' | +# .-------. | .------------. .------------. .-------. | +# | ospfd | | | p0 | net_tap | x-p0 | | zebra | | +# '-------' | | +-------------+ | '-------' | +# .------. | 172.16.0.1 | | 172.16.0.2 |.----------. | +# | main | '------------' '------------'| lo | | +# '------' | | | | | +# | ping <------------------------------------> | 16.0.0.1 | | +# | | | '----------' | +# '-----------' '--------------------' + +. $(dirname $0)/_init_frr.sh + +create_interface p0 +set_ip_address p0 172.16.0.1/24 + +start_frr ospf-peer 0 +move_to_netns x-p0 ospf-peer + +# Enable kernel MPLS in the peer netns so that FRR zebra can install the +# pop rule for its own node SID (label 16001) when OSPF-SR comes up. +ip netns exec ospf-peer sysctl -wq net.mpls.platform_labels=17000 +ip netns exec ospf-peer sysctl -wq net.mpls.conf.x-p0.input=1 + +# Configure grout's FRR: OSPF with Segment Routing +vtysh <<-EOF +configure terminal +ip router-id 172.16.0.1 +! +interface lo + ip address 17.0.0.1/32 +exit +! +interface p0 + ip ospf hello-interval 1 + ip ospf network point-to-point +exit +! +router ospf + ospf router-id 172.16.0.1 + network 172.16.0.0/24 area 0 + network 17.0.0.1/32 area 0 + router-info area + segment-routing global-block 16000 23999 + segment-routing on + segment-routing node-msd 8 + segment-routing prefix 17.0.0.1/32 index 0 +exit +! +EOF + +vtysh -N ospf-peer <<-EOF +configure terminal +ip router-id 172.16.0.2 +! +interface lo + ip address 16.0.0.1/32 +exit +! +interface x-p0 + ip address 172.16.0.2/24 + ip ospf hello-interval 1 + ip ospf network point-to-point +exit +! +router ospf + ospf router-id 172.16.0.2 + network 172.16.0.0/24 area 0 + network 16.0.0.1/32 area 0 + router-info area + segment-routing global-block 16000 23999 + segment-routing on + segment-routing node-msd 8 + segment-routing prefix 16.0.0.1/32 index 1 +exit +! +EOF + +attempts=60 +while ! vtysh -c 'show ip ospf neighbor json' | jq -e '.neighbors."172.16.0.2"[0].converged == "Full"'; do + sleep 1 + if [ "$attempts" -le 0 ]; then + fail "OSPF failed to connect to neighbor." + fi + attempts=$((attempts - 1)) +done + +# Wait for OSPF route exchange +attempts=30 +while ! vtysh -c 'show ip route ospf json' | jq -e '."16.0.0.1/32"'; do + sleep 1 + if [ "$attempts" -le 0 ]; then + fail "OSPF failed to get routes." + fi + attempts=$((attempts - 1)) +done + +# OSPF-SR installs LSP for peer's node SID in grout LFIB +wait_event -t 60 'mpls route add: vrf=main label=1[56][0-9][0-9][0-9] .* origin=ospf' + +# IP route is installed by dplane_grout synchronously with the RIB update above +grcli ping 16.0.0.1 count 1