diff --git a/Documentation/netlink/specs/rt-route.yaml b/Documentation/netlink/specs/rt-route.yaml index 33195db96746..253037ea5176 100644 --- a/Documentation/netlink/specs/rt-route.yaml +++ b/Documentation/netlink/specs/rt-route.yaml @@ -78,6 +78,27 @@ definitions: - name: rta-used type: u32 + - + name: del-reason + type: enum + name-prefix: rt-del-reason- + enum-name: rt-del-reason + doc: | + Why the kernel deleted a route. New causes may be appended. The + value space is family-agnostic, a value is never reinterpreted + per address family. + entries: + - + name: unspec + doc: The deletion path does not record a cause. + - + name: expired + doc: The RTF_EXPIRES lifetime ran out and the route was garbage + collected. + - + name: ra-withdrawn + doc: A Router Advertisement withdrew the route with a zero + lifetime. attribute-sets: - @@ -185,6 +206,18 @@ attribute-sets: type: u32 byte-order: big-endian display-hint: hex + - + name: del-reason + type: u32 + enum: del-reason + doc: | + Emitted only on RTM_DELROUTE notifications, and only when the + deletion path records a cause. An absent attribute means + either an older kernel or a deletion path that does not + record its cause, consumers must treat absent and unspec + identically. Currently only IPv6 deletion paths record a + cause. The attribute is notification-only, the kernel rejects + it in requests. - name: metrics name-prefix: rtax- @@ -299,6 +332,7 @@ operations: - dport - nh-id - flowlabel + - del-reason dump: request: value: 26 @@ -313,7 +347,35 @@ operations: do: request: value: 24 - attributes: *all-route-attrs + attributes: &route-req-attrs + - dst + - src + - iif + - oif + - gateway + - priority + - prefsrc + - metrics + - multipath + - flow + - cacheinfo + - table + - mark + - mfc-stats + - via + - newdst + - pref + - encap-type + - encap + - expires + - pad + - uid + - ttl-propagate + - ip-proto + - sport + - dport + - nh-id + - flowlabel - name: delroute doc: Delete an existing route @@ -321,4 +383,23 @@ operations: do: request: value: 25 - attributes: *all-route-attrs + attributes: *route-req-attrs + - + name: newroute-ntf + doc: Notification about a created route. + value: 24 + notify: getroute + - + name: delroute-ntf + doc: Notification about a deleted route. + value: 25 + notify: getroute + +mcast-groups: + list: + - + name: rtnlgrp-ipv4-route + value: 7 + - + name: rtnlgrp-ipv6-route + value: 11 diff --git a/include/net/ip6_fib.h b/include/net/ip6_fib.h index 9cd27e1b9b69..232289b439c3 100644 --- a/include/net/ip6_fib.h +++ b/include/net/ip6_fib.h @@ -468,7 +468,8 @@ void fib6_clean_all_skip_notify(struct net *net, int fib6_add(struct fib6_node *root, struct fib6_info *rt, struct nl_info *info, struct netlink_ext_ack *extack); -int fib6_del(struct fib6_info *rt, struct nl_info *info); +int fib6_del(struct fib6_info *rt, struct nl_info *info, + enum rt_del_reason del_reason); static inline void rt6_get_prefsrc(const struct rt6_info *rt, struct in6_addr *addr) @@ -532,6 +533,8 @@ static inline void fib6_rt_update(struct net *net, struct fib6_info *rt, #endif void inet6_rt_notify(int event, struct fib6_info *rt, struct nl_info *info, unsigned int flags); +void inet6_rt_del_notify(struct fib6_info *rt, struct nl_info *info, + enum rt_del_reason del_reason); void fib6_age_exceptions(struct fib6_info *rt, struct fib6_gc_args *gc_args, unsigned long now); diff --git a/include/net/ip6_route.h b/include/net/ip6_route.h index 09ffe0f13ce7..cc045705862d 100644 --- a/include/net/ip6_route.h +++ b/include/net/ip6_route.h @@ -126,6 +126,8 @@ int ipv6_route_ioctl(struct net *net, unsigned int cmd, int ip6_route_add(struct fib6_config *cfg, gfp_t gfp_flags, struct netlink_ext_ack *extack); int ip6_ins_rt(struct net *net, struct fib6_info *f6i); +int ip6_del_rt_reason(struct net *net, struct fib6_info *f6i, + enum rt_del_reason del_reason); #if IS_ENABLED(CONFIG_IPV6) int ip6_del_rt(struct net *net, struct fib6_info *f6i, bool skip_notify); #else diff --git a/include/uapi/linux/rtnetlink.h b/include/uapi/linux/rtnetlink.h index 27265fd31e5f..3988b51db539 100644 --- a/include/uapi/linux/rtnetlink.h +++ b/include/uapi/linux/rtnetlink.h @@ -399,6 +399,7 @@ enum rtattr_type_t { RTA_DPORT, RTA_NH_ID, RTA_FLOWLABEL, + RTA_DEL_REASON, __RTA_MAX }; @@ -407,6 +408,30 @@ enum rtattr_type_t { #define RTM_RTA(r) ((struct rtattr*)(((char*)(r)) + NLMSG_ALIGN(sizeof(struct rtmsg)))) #define RTM_PAYLOAD(n) NLMSG_PAYLOAD(n,sizeof(struct rtmsg)) +/* RTA_DEL_REASON: why the kernel deleted the route. u32. + * Emitted only on RTM_DELROUTE notifications, and only when the deletion + * path records a cause. Absence means either an older kernel or a + * deletion path that does not (yet) record its cause - consumers must + * treat "absent" and "unspec" identically. New causes may be appended. + * Currently only IPv6 deletion paths record a cause. + * + * The attribute is notification-only: the kernel rejects it in + * requests, so a notification must not be echoed back verbatim. + * + * The value space is family-agnostic: a value must never be + * reinterpreted per address family. A cause that only one family can + * produce still gets its own value rather than reusing another + * family's. + */ +enum rt_del_reason { + RT_DEL_REASON_UNSPEC, /* cause not recorded */ + RT_DEL_REASON_EXPIRED, /* RTF_EXPIRES lifetime ran out (GC) */ + RT_DEL_REASON_RA_WITHDRAWN, /* zero-lifetime RA / PIO / RIO */ + __RT_DEL_REASON_MAX +}; + +#define RT_DEL_REASON_MAX (__RT_DEL_REASON_MAX - 1) + /* RTM_MULTIPATH --- array of struct rtnexthop. * * "struct rtnexthop" describes all necessary nexthop information, diff --git a/net/ipv6/addrconf.c b/net/ipv6/addrconf.c index f6fa2715b450..9d89be7e0544 100644 --- a/net/ipv6/addrconf.c +++ b/net/ipv6/addrconf.c @@ -2874,7 +2874,8 @@ void addrconf_prefix_rcv(struct net_device *dev, u8 *opt, int len, bool sllao) if (rt) { /* Autoconf prefix route */ if (valid_lft == 0) { - ip6_del_rt(net, rt, false); + ip6_del_rt_reason(net, rt, + RT_DEL_REASON_RA_WITHDRAWN); rt = NULL; } else { table = rt->fib6_table; diff --git a/net/ipv6/ip6_fib.c b/net/ipv6/ip6_fib.c index e9fc692d4f3b..3e382ba1573e 100644 --- a/net/ipv6/ip6_fib.c +++ b/net/ipv6/ip6_fib.c @@ -1967,7 +1967,8 @@ static struct fib6_node *fib6_repair_tree(struct net *net, } static void fib6_del_route(struct fib6_table *table, struct fib6_node *fn, - struct fib6_info __rcu **rtp, struct nl_info *info) + struct fib6_info __rcu **rtp, struct nl_info *info, + enum rt_del_reason del_reason) { struct fib6_info *leaf, *replace_rt = NULL; struct fib6_walker *w; @@ -2056,13 +2057,14 @@ static void fib6_del_route(struct fib6_table *table, struct fib6_node *fn, call_fib6_entry_notifiers_replace(net, replace_rt); } if (!info->skip_notify) - inet6_rt_notify(RTM_DELROUTE, rt, info, 0); + inet6_rt_del_notify(rt, info, del_reason); fib6_info_release(rt); } /* Need to own table->tb6_lock */ -int fib6_del(struct fib6_info *rt, struct nl_info *info) +int fib6_del(struct fib6_info *rt, struct nl_info *info, + enum rt_del_reason del_reason) { struct net *net = info->nl_net; struct fib6_info __rcu **rtp; @@ -2091,7 +2093,7 @@ int fib6_del(struct fib6_info *rt, struct nl_info *info) if (rt == cur) { if (fib6_requires_src(cur)) fib6_routes_require_src_dec(info->nl_net); - fib6_del_route(table, fn, rtp, info); + fib6_del_route(table, fn, rtp, info, del_reason); return 0; } rtp_next = &cur->fib6_next; @@ -2253,7 +2255,7 @@ static int fib6_clean_node(struct fib6_walker *w) res = c->func(rt, c->arg); if (res == -1) { w->leaf = rt; - res = fib6_del(rt, &info); + res = fib6_del(rt, &info, RT_DEL_REASON_UNSPEC); if (res) { #if RT6_DEBUG >= 2 pr_debug("%s: del failed: rt=%p@%p err=%d\n", @@ -2401,7 +2403,7 @@ static void fib6_gc_table(struct net *net, hlist_for_each_entry_safe(rt, n, &tb6->tb6_gc_hlist, gc_link) if (fib6_age(rt, gc_args) == -1) - fib6_del(rt, &info); + fib6_del(rt, &info, RT_DEL_REASON_EXPIRED); } static void fib6_gc_all(struct net *net, struct fib6_gc_args *gc_args) diff --git a/net/ipv6/ndisc.c b/net/ipv6/ndisc.c index 2ceb655c4229..75515fd99383 100644 --- a/net/ipv6/ndisc.c +++ b/net/ipv6/ndisc.c @@ -1362,7 +1362,9 @@ static enum skb_drop_reason ndisc_router_discovery(struct sk_buff *skb) defrtr_usr_metric = in6_dev->cnf.ra_defrtr_metric; /* delete the route if lifetime is 0 or if metric needs change */ if (rt && (lifetime == 0 || rt->fib6_metric != defrtr_usr_metric)) { - ip6_del_rt(net, rt, false); + ip6_del_rt_reason(net, rt, + lifetime == 0 ? RT_DEL_REASON_RA_WITHDRAWN : + RT_DEL_REASON_UNSPEC); rt = NULL; } diff --git a/net/ipv6/route.c b/net/ipv6/route.c index 5968ce5ad150..ae2f93a19dd9 100644 --- a/net/ipv6/route.c +++ b/net/ipv6/route.c @@ -111,7 +111,7 @@ static int rt6_fill_node(struct net *net, struct sk_buff *skb, struct fib6_info *rt, struct dst_entry *dst, struct in6_addr *dest, struct in6_addr *src, int iif, int type, u32 portid, u32 seq, - unsigned int flags); + unsigned int flags, enum rt_del_reason del_reason); static struct rt6_info *rt6_find_cached_rt(const struct fib6_result *res, const struct in6_addr *daddr, const struct in6_addr *saddr); @@ -1020,7 +1020,7 @@ int rt6_route_rcv(struct net_device *dev, u8 *opt, int len, gwaddr, dev); if (rt && !lifetime) { - ip6_del_rt(net, rt, false); + ip6_del_rt_reason(net, rt, RT_DEL_REASON_RA_WITHDRAWN); rt = NULL; } @@ -3973,7 +3973,8 @@ int ip6_route_add(struct fib6_config *cfg, gfp_t gfp_flags, return err; } -static int __ip6_del_rt(struct fib6_info *rt, struct nl_info *info) +static int __ip6_del_rt(struct fib6_info *rt, struct nl_info *info, + enum rt_del_reason del_reason) { struct net *net = info->nl_net; struct fib6_table *table; @@ -3986,7 +3987,7 @@ static int __ip6_del_rt(struct fib6_info *rt, struct nl_info *info) table = rt->fib6_table; spin_lock_bh(&table->tb6_lock); - err = fib6_del(rt, info); + err = fib6_del(rt, info, del_reason); spin_unlock_bh(&table->tb6_lock); out: @@ -4001,7 +4002,15 @@ int ip6_del_rt(struct net *net, struct fib6_info *rt, bool skip_notify) .skip_notify = skip_notify }; - return __ip6_del_rt(rt, &info); + return __ip6_del_rt(rt, &info, RT_DEL_REASON_UNSPEC); +} + +int ip6_del_rt_reason(struct net *net, struct fib6_info *rt, + enum rt_del_reason del_reason) +{ + struct nl_info info = { .nl_net = net }; + + return __ip6_del_rt(rt, &info, del_reason); } static int __ip6_del_rt_siblings(struct fib6_info *rt, struct fib6_config *cfg) @@ -4028,7 +4037,8 @@ static int __ip6_del_rt_siblings(struct fib6_info *rt, struct fib6_config *cfg) if (rt6_fill_node(net, skb, rt, NULL, NULL, NULL, 0, RTM_DELROUTE, - info->portid, seq, 0) < 0) { + info->portid, seq, 0, + RT_DEL_REASON_UNSPEC) < 0) { kfree_skb(skb); skb = NULL; } else @@ -4064,13 +4074,13 @@ static int __ip6_del_rt_siblings(struct fib6_info *rt, struct fib6_config *cfg) list_for_each_entry_safe(sibling, next_sibling, &rt->fib6_siblings, fib6_siblings) { - err = fib6_del(sibling, info); + err = fib6_del(sibling, info, RT_DEL_REASON_UNSPEC); if (err) goto out_unlock; } } - err = fib6_del(rt, info); + err = fib6_del(rt, info, RT_DEL_REASON_UNSPEC); out_unlock: spin_unlock_bh(&table->tb6_lock); out_put: @@ -4196,7 +4206,8 @@ static int ip6_route_del(struct fib6_config *cfg, if (!fib6_info_hold_safe(rt)) continue; - err = __ip6_del_rt(rt, &cfg->fc_nlinfo); + err = __ip6_del_rt(rt, &cfg->fc_nlinfo, + RT_DEL_REASON_UNSPEC); break; } if (cfg->fc_nh_id) @@ -4215,7 +4226,8 @@ static int ip6_route_del(struct fib6_config *cfg, /* if gateway was specified only delete the one hop */ if (cfg->fc_flags & RTF_GATEWAY) - err = __ip6_del_rt(rt, &cfg->fc_nlinfo); + err = __ip6_del_rt(rt, &cfg->fc_nlinfo, + RT_DEL_REASON_UNSPEC); else err = __ip6_del_rt_siblings(rt, cfg); break; @@ -5739,6 +5751,7 @@ static size_t rt6_nlmsg_size(struct fib6_info *f6i) + nla_total_size(sizeof(struct rta_cacheinfo)) + nla_total_size(TCP_CA_NAME_MAX) /* RTAX_CC_ALGO */ + nla_total_size(1) /* RTA_PREF */ + + nla_total_size(4) /* RTA_DEL_REASON */ + nexthop_len; } @@ -5775,7 +5788,7 @@ static int rt6_fill_node(struct net *net, struct sk_buff *skb, struct fib6_info *rt, struct dst_entry *dst, struct in6_addr *dest, struct in6_addr *src, int iif, int type, u32 portid, u32 seq, - unsigned int flags) + unsigned int flags, enum rt_del_reason del_reason) { struct rt6_info *rt6 = dst_rt6_info(dst); struct rt6key *rt6_dst, *rt6_src; @@ -5954,6 +5967,10 @@ static int rt6_fill_node(struct net *net, struct sk_buff *skb, if (rtnl_put_cacheinfo(skb, dst, 0, expires, dst ? dst->error : 0) < 0) goto nla_put_failure; + if (del_reason != RT_DEL_REASON_UNSPEC && + nla_put_u32(skb, RTA_DEL_REASON, del_reason)) + goto nla_put_failure; + if (nla_put_u8(skb, RTA_PREF, IPV6_EXTRACT_PREF(rt6_flags))) goto nla_put_failure; @@ -6055,7 +6072,8 @@ static int rt6_nh_dump_exceptions(struct fib6_nh *nh, void *arg) &rt6_ex->rt6i->dst, NULL, NULL, 0, RTM_NEWROUTE, NETLINK_CB(dump->cb->skb).portid, - dump->cb->nlh->nlmsg_seq, w->flags); + dump->cb->nlh->nlmsg_seq, w->flags, + RT_DEL_REASON_UNSPEC); if (err) return err; @@ -6103,7 +6121,8 @@ int rt6_dump_route(struct fib6_info *rt, void *p_arg, unsigned int skip) if (rt6_fill_node(net, arg->skb, rt, NULL, NULL, NULL, 0, RTM_NEWROUTE, NETLINK_CB(arg->cb->skb).portid, - arg->cb->nlh->nlmsg_seq, flags)) { + arg->cb->nlh->nlmsg_seq, flags, + RT_DEL_REASON_UNSPEC)) { return 0; } count++; @@ -6336,12 +6355,14 @@ static int inet6_rtm_getroute(struct sk_buff *in_skb, struct nlmsghdr *nlh, err = rt6_fill_node(net, skb, from, NULL, NULL, NULL, iif, RTM_NEWROUTE, NETLINK_CB(in_skb).portid, - nlh->nlmsg_seq, 0); + nlh->nlmsg_seq, 0, + RT_DEL_REASON_UNSPEC); else err = rt6_fill_node(net, skb, from, dst, &fl6.daddr, &fl6.saddr, iif, RTM_NEWROUTE, NETLINK_CB(in_skb).portid, - nlh->nlmsg_seq, 0); + nlh->nlmsg_seq, 0, + RT_DEL_REASON_UNSPEC); } else { err = -ENETUNREACH; } @@ -6357,8 +6378,9 @@ static int inet6_rtm_getroute(struct sk_buff *in_skb, struct nlmsghdr *nlh, return err; } -void inet6_rt_notify(int event, struct fib6_info *rt, struct nl_info *info, - unsigned int nlm_flags) +static void __inet6_rt_notify(int event, struct fib6_info *rt, + struct nl_info *info, unsigned int nlm_flags, + enum rt_del_reason del_reason) { struct net *net = info->nl_net; struct sk_buff *skb; @@ -6377,7 +6399,7 @@ void inet6_rt_notify(int event, struct fib6_info *rt, struct nl_info *info, goto errout; err = rt6_fill_node(net, skb, rt, NULL, NULL, NULL, 0, - event, info->portid, seq, nlm_flags); + event, info->portid, seq, nlm_flags, del_reason); if (err < 0) { kfree_skb(skb); /* -EMSGSIZE implies needed space grew under us. */ @@ -6398,6 +6420,18 @@ void inet6_rt_notify(int event, struct fib6_info *rt, struct nl_info *info, rtnl_set_sk_err(net, RTNLGRP_IPV6_ROUTE, err); } +void inet6_rt_notify(int event, struct fib6_info *rt, struct nl_info *info, + unsigned int nlm_flags) +{ + __inet6_rt_notify(event, rt, info, nlm_flags, RT_DEL_REASON_UNSPEC); +} + +void inet6_rt_del_notify(struct fib6_info *rt, struct nl_info *info, + enum rt_del_reason del_reason) +{ + __inet6_rt_notify(RTM_DELROUTE, rt, info, 0, del_reason); +} + void fib6_rt_update(struct net *net, struct fib6_info *rt, struct nl_info *info) { @@ -6410,7 +6444,8 @@ void fib6_rt_update(struct net *net, struct fib6_info *rt, goto errout; err = rt6_fill_node(net, skb, rt, NULL, NULL, NULL, 0, - RTM_NEWROUTE, info->portid, seq, NLM_F_REPLACE); + RTM_NEWROUTE, info->portid, seq, NLM_F_REPLACE, + RT_DEL_REASON_UNSPEC); if (err < 0) { /* -EMSGSIZE implies BUG in rt6_nlmsg_size() */ WARN_ON(err == -EMSGSIZE); @@ -6463,7 +6498,7 @@ void fib6_info_hw_flags_set(struct net *net, struct fib6_info *f6i, } err = rt6_fill_node(net, skb, f6i, NULL, NULL, NULL, 0, RTM_NEWROUTE, 0, - 0, 0); + 0, 0, RT_DEL_REASON_UNSPEC); if (err < 0) { /* -EMSGSIZE implies BUG in rt6_nlmsg_size() */ WARN_ON(err == -EMSGSIZE); diff --git a/tools/testing/selftests/net/lib/py/__init__.py b/tools/testing/selftests/net/lib/py/__init__.py index e58bdbdc58ee..34935886b6ad 100644 --- a/tools/testing/selftests/net/lib/py/__init__.py +++ b/tools/testing/selftests/net/lib/py/__init__.py @@ -17,7 +17,7 @@ from .utils import CmdExitFailure, fd_read_timeout, cmd, bkg, defer, \ wait_file, tool, tc from .bpf import bpf_map_set, bpf_map_dump, bpf_prog_map_ids from .ynl import NlError, NlctrlFamily, YnlFamily, \ - EthtoolFamily, NetdevFamily, RtnlFamily, RtnlAddrFamily + EthtoolFamily, NetdevFamily, RtnlFamily, RtnlAddrFamily, RtnlRouteFamily from .ynl import NetshaperFamily, DevlinkFamily, PSPFamily, Netlink __all__ = ["KSRC", @@ -34,4 +34,4 @@ __all__ = ["KSRC", "NetdevSim", "NetdevSimDev", "NetshaperFamily", "DevlinkFamily", "PSPFamily", "NlError", "YnlFamily", "EthtoolFamily", "NetdevFamily", "RtnlFamily", - "NlctrlFamily", "RtnlAddrFamily", "Netlink"] + "NlctrlFamily", "RtnlAddrFamily", "RtnlRouteFamily", "Netlink"] diff --git a/tools/testing/selftests/net/lib/py/ynl.py b/tools/testing/selftests/net/lib/py/ynl.py index 2e567062aa6c..08deff756f29 100644 --- a/tools/testing/selftests/net/lib/py/ynl.py +++ b/tools/testing/selftests/net/lib/py/ynl.py @@ -29,7 +29,7 @@ except ModuleNotFoundError as e: __all__ = [ "NlError", "NlPolicy", "Netlink", "YnlFamily", "SPEC_PATH", - "EthtoolFamily", "RtnlFamily", "RtnlAddrFamily", + "EthtoolFamily", "RtnlFamily", "RtnlAddrFamily", "RtnlRouteFamily", "NetdevFamily", "NetshaperFamily", "NlctrlFamily", "DevlinkFamily", "PSPFamily", ] @@ -54,6 +54,11 @@ class RtnlAddrFamily(YnlFamily): super().__init__((SPEC_PATH / Path('rt-addr.yaml')).as_posix(), schema='', recv_size=recv_size) +class RtnlRouteFamily(YnlFamily): + def __init__(self, recv_size=0): + super().__init__((SPEC_PATH / Path('rt-route.yaml')).as_posix(), + schema='', recv_size=recv_size) + class NetdevFamily(YnlFamily): def __init__(self, recv_size=0): super().__init__((SPEC_PATH / Path('netdev.yaml')).as_posix(), diff --git a/tools/testing/selftests/net/rtnetlink.py b/tools/testing/selftests/net/rtnetlink.py index 0c67c7c00d84..5cc3ebdcf08d 100755 --- a/tools/testing/selftests/net/rtnetlink.py +++ b/tools/testing/selftests/net/rtnetlink.py @@ -5,7 +5,9 @@ import socket import struct import time from lib.py import bkg, ip, ksft_exit, ksft_run, ksft_eq, ksft_ge, ksft_true, KsftSkipEx -from lib.py import CmdExitFailure, NetNS, NetNSEnter, RtnlAddrFamily +from lib.py import ksft_not_in, ksft_not_none +from lib.py import CmdExitFailure, NetNS, NetNSEnter, RtnlAddrFamily, RtnlRouteFamily +from lib.py import defer IPV4_ALL_HOSTS_MULTICAST = b'\xe0\x00\x00\x01' IPV4_TEST_MULTICAST = b'\xef\x01\x01\x01' @@ -134,8 +136,189 @@ def ipv4_devconf_notify() -> None: ksft_true(f"inet {ifname} forwarding on" in cmd_obj.stdout, f"No 'forwarding on' notificiation found for interface {ifname}") +def _rtnl_route_subscribe(ns): + with NetNSEnter(str(ns)): + rtnl = RtnlRouteFamily() + defer(rtnl.close) + rtnl.ntf_subscribe("rtnlgrp-ipv6-route") + return rtnl + + +def _wait_route_ntf(rtnl, name, dst_len, dst=None, deadline=10): + """Return the attrs of the first matching notification, None on timeout.""" + + for msg in rtnl.poll_ntf(duration=deadline): + if msg['name'] != name: + continue + attrs = msg['msg'] + if attrs['rtm-dst-len'] != dst_len: + continue + if dst is not None and attrs.get('dst') != dst: + continue + return attrs + return None + + +def _collect_route_ntfs(rtnl, name, want, deadline=10): + """Gather attrs of matching notifications, keyed by (dst_len, dst).""" + + seen = {} + for msg in rtnl.poll_ntf(duration=deadline): + if msg['name'] != name: + continue + attrs = msg['msg'] + key = (attrs['rtm-dst-len'], attrs.get('dst')) + if key in want: + seen[key] = attrs + if len(seen) == len(want): + break + return seen + + +def _write_ipv6_sysctl(name, value): + with open(f"/proc/sys/net/ipv6/{name}", "w") as f: + f.write(f"{value}\n") + + +def ipv6_route_del_reason_expired() -> None: + """An expired route reports RTA_DEL_REASON == expired.""" + + with NetNS() as ns: + rtnl = _rtnl_route_subscribe(ns) + with NetNSEnter(str(ns)): + _write_ipv6_sysctl("route/gc_interval", 2) + ip("link add name dummy1 type dummy", ns=str(ns)) + ip("link set dev dummy1 up", ns=str(ns)) + ip("-6 route add 2001:db8:2::/64 dev dummy1 expires 2", ns=str(ns)) + + attrs = _wait_route_ntf(rtnl, 'delroute-ntf', 64, '2001:db8:2::', + deadline=15) + ksft_not_none(attrs, "no RTM_DELROUTE for the expired route") + if attrs is not None: + ksft_eq(attrs.get('del-reason'), 'expired') + + +def _send_ra(sock, ifindex, lifetime, rio=None, pio=None): + """The kernel fills in the ICMPv6 checksum on raw ICMPv6 sockets.""" + + # type, code, cksum, hop limit, flags, router lifetime, + # reachable time, retrans timer + ra = struct.pack('!BBHBBHII', 134, 0, 0, 64, 0, lifetime, 0, 0) + if rio is not None: + prefix, plen, rio_lifetime = rio + # RFC 4191 route information option, /64 prefix (8 bytes) + ra += struct.pack('!BBBBI', 24, 2, plen, 0, rio_lifetime) + ra += socket.inet_pton(socket.AF_INET6, prefix)[:8] + if pio is not None: + prefix, plen, valid_lft = pio + # RFC 4861 prefix information option, on-link only (L set, A clear) + ra += struct.pack('!BBBBIII', 3, 4, plen, 0x80, valid_lft, 0, 0) + ra += socket.inet_pton(socket.AF_INET6, prefix) + sock.sendto(ra, ('ff02::1', 0, 0, ifindex)) + + +def _ra_router_sock(ns_r, ifname): + with NetNSEnter(str(ns_r)): + sock = socket.socket(socket.AF_INET6, socket.SOCK_RAW, + socket.IPPROTO_ICMPV6) + sock.setsockopt(socket.IPPROTO_IPV6, socket.IPV6_MULTICAST_HOPS, 255) + defer(sock.close) + return sock, socket.if_nametoindex(ifname) + + +def _ra_advertise_routes(rtnl, sock, ifindex, want, **ra_opts): + """ + Sending fails with EADDRNOTAVAIL while the router's link-local + address is still tentative. addrconf_dad_start() only queues + addrconf_dad_work(), and IFA_F_TENTATIVE is cleared when that work + item runs, so retry until it does. + """ + + seen = {} + for _ in range(10): + try: + _send_ra(sock, ifindex, **ra_opts) + except OSError: + time.sleep(0.2) + continue + seen.update(_collect_route_ntfs(rtnl, 'newroute-ntf', + want - set(seen.keys()), deadline=2)) + if len(seen) == len(want): + break + return seen + + +def ipv6_route_del_reason_ra_withdrawn() -> None: + """ + Routes withdrawn by a zero-lifetime RA (router lifetime, RFC 4861 + PIO, RFC 4191 RIO) report RTA_DEL_REASON == ra-withdrawn. + """ + + # (rtm-dst-len, dst); the default route carries no RTA_DST + routes = {(0, None), (64, '2001:db8:6::'), (64, '2001:db8:5::')} + + with NetNS() as ns_h, NetNS() as ns_r: + ip(f"link add veth0 netns {ns_h} type veth peer name veth1 netns {ns_r}") + with NetNSEnter(str(ns_h)): + _write_ipv6_sysctl("conf/veth0/accept_ra", 2) + _write_ipv6_sysctl("conf/veth0/forwarding", 0) + try: + _write_ipv6_sysctl("conf/veth0/accept_ra_rt_info_max_plen", 64) + except FileNotFoundError: + raise KsftSkipEx("no CONFIG_IPV6_ROUTE_INFO") + with NetNSEnter(str(ns_r)): + # skip the DAD probe so the router's link-local source only + # has to wait for addrconf_dad_work() to clear IFA_F_TENTATIVE + _write_ipv6_sysctl("conf/veth1/accept_dad", 0) + ip("link set dev veth0 up", ns=str(ns_h)) + ip("link set dev veth1 up", ns=str(ns_r)) + + rtnl = _rtnl_route_subscribe(ns_h) + sock, ifindex = _ra_router_sock(ns_r, "veth1") + + seen = _ra_advertise_routes(rtnl, sock, ifindex, routes, + lifetime=1800, + rio=('2001:db8:5::', 64, 600), + pio=('2001:db8:6::', 64, 600)) + ksft_eq(set(seen), routes, "not all RA routes were installed") + if set(seen) != routes: + return + + _send_ra(sock, ifindex, 0, rio=('2001:db8:5::', 64, 0), + pio=('2001:db8:6::', 64, 0)) + seen = _collect_route_ntfs(rtnl, 'delroute-ntf', routes) + for key in routes: + attrs = seen.get(key) + ksft_not_none(attrs, f"no RTM_DELROUTE for {key}") + if attrs is not None: + ksft_eq(attrs.get('del-reason'), 'ra-withdrawn') + + +def ipv6_route_del_reason_absent() -> None: + """ + A deletion path that records no cause (here a userspace request) + must not carry RTA_DEL_REASON at all. + """ + + with NetNS() as ns: + rtnl = _rtnl_route_subscribe(ns) + ip("link add name dummy1 type dummy", ns=str(ns)) + ip("link set dev dummy1 up", ns=str(ns)) + ip("-6 route add 2001:db8:1::/64 dev dummy1", ns=str(ns)) + ip("-6 route del 2001:db8:1::/64 dev dummy1", ns=str(ns)) + + attrs = _wait_route_ntf(rtnl, 'delroute-ntf', 64, '2001:db8:1::') + ksft_not_none(attrs, "no RTM_DELROUTE for 2001:db8:1::/64") + if attrs is not None: + ksft_not_in('del-reason', attrs, + "user deletion must not carry del-reason") + + def main() -> None: - ksft_run([dump_mcaddr_check, dump_mcaddr6_check, ipv4_devconf_notify]) + ksft_run([dump_mcaddr_check, dump_mcaddr6_check, ipv4_devconf_notify, + ipv6_route_del_reason_expired, + ipv6_route_del_reason_ra_withdrawn, + ipv6_route_del_reason_absent]) ksft_exit() if __name__ == "__main__":