include/net/dropreason-core.h | 6 ++++++ net/ipv4/ip_forward.c | 2 +- net/ipv6/exthdrs.c | 6 +++--- net/ipv6/ip6_output.c | 2 +- 4 files changed, 11 insertions(+), 5 deletions(-)
The forwarding paths report an expired TTL or hop limit as
SKB_DROP_REASON_IP_INHDR, the reason otherwise used for a header that is
malformed (ip_input.c, exthdrs.c, br_netfilter). Nothing else in the drop
path separates the two: IPSTATS_MIB_INHDRERRORS covers both, and the TTL
check runs before NF_INET_FORWARD, so netfilter tracing stops at
PREROUTING and never sees the drop.
The Fedora bug linked below shows how that reads in practice. The
reporter took kfree_skb(reason=IP_INHDR, loc=ip_forward) to mean the
software header checksum check had failed, and worked through RX checksum
offload, tc csum actions and both libvirt firewall backends before the
drops turned out to be replies arriving with TTL 1. ip_forward() never
verifies the header checksum; that runs earlier, in ip_rcv_core(), and
reports IP_CSUM.
TTL expiry is not a corner case -- every traceroute through a Linux
router goes through too_many_hops.
The three loopback hop limit checks in exthdrs.c drop with no reason at
all; give them the new one.
IPSTATS_MIB_INHDRERRORS stays as it is: RFC 1213 counts time-to-live
exceeded under ipInHdrErrors. The drop reason has no such constraint.
Link: https://bugzilla.redhat.com/show_bug.cgi?id=2517131
Signed-off-by: Junjie Cao <junjie.cao@intel.com>
---
v3: drop the "on a packet being forwarded" qualifier from the kernel-doc;
the reason also covers the loopback hop-limit checks in exthdrs.c
(David Ahern)
v2: https://lore.kernel.org/netdev/20260901020613.417495-1-junjie.cao@intel.com/
v1: https://lore.kernel.org/netdev/20260825073906.336072-1-junjie.cao@intel.com/
include/net/dropreason-core.h | 6 ++++++
net/ipv4/ip_forward.c | 2 +-
net/ipv6/exthdrs.c | 6 +++---
net/ipv6/ip6_output.c | 2 +-
4 files changed, 11 insertions(+), 5 deletions(-)
diff --git a/include/net/dropreason-core.h b/include/net/dropreason-core.h
index 2f312d1f67d6..ed2c02884b68 100644
--- a/include/net/dropreason-core.h
+++ b/include/net/dropreason-core.h
@@ -128,6 +128,7 @@
FN(PSP_INPUT) \
FN(PSP_OUTPUT) \
FN(RECURSION_LIMIT) \
+ FN(IP_TTL_EXCEEDED) \
FNe(MAX)
/**
@@ -606,6 +607,11 @@ enum skb_drop_reason {
SKB_DROP_REASON_PSP_OUTPUT,
/** @SKB_DROP_REASON_RECURSION_LIMIT: Dead loop on virtual device. */
SKB_DROP_REASON_RECURSION_LIMIT,
+ /**
+ * @SKB_DROP_REASON_IP_TTL_EXCEEDED: IPv4 TTL or IPv6 hop limit hit
+ * zero (see IPSTATS_MIB_INHDRERRORS)
+ */
+ SKB_DROP_REASON_IP_TTL_EXCEEDED,
/**
* @SKB_DROP_REASON_MAX: the maximum of core drop reasons, which
* shouldn't be used as a real 'reason' - only for tracing code gen
diff --git a/net/ipv4/ip_forward.c b/net/ipv4/ip_forward.c
index 8b65f12583eb..b242561d37e7 100644
--- a/net/ipv4/ip_forward.c
+++ b/net/ipv4/ip_forward.c
@@ -174,7 +174,7 @@ int ip_forward(struct sk_buff *skb)
/* Tell the sender its packet died... */
__IP_INC_STATS(net, IPSTATS_MIB_INHDRERRORS);
icmp_send(skb, ICMP_TIME_EXCEEDED, ICMP_EXC_TTL, 0);
- SKB_DR_SET(reason, IP_INHDR);
+ SKB_DR_SET(reason, IP_TTL_EXCEEDED);
drop:
kfree_skb_reason(skb, reason);
return NET_RX_DROP;
diff --git a/net/ipv6/exthdrs.c b/net/ipv6/exthdrs.c
index 51941ad656a3..74caaf8746ff 100644
--- a/net/ipv6/exthdrs.c
+++ b/net/ipv6/exthdrs.c
@@ -464,7 +464,7 @@ static int ipv6_srh_rcv(struct sk_buff *skb, struct inet6_dev *idev)
__IP6_INC_STATS(net, idev, IPSTATS_MIB_INHDRERRORS);
icmpv6_send(skb, ICMPV6_TIME_EXCEED,
ICMPV6_EXC_HOPLIMIT, 0);
- kfree_skb(skb);
+ kfree_skb_reason(skb, SKB_DROP_REASON_IP_TTL_EXCEEDED);
return -1;
}
ipv6_hdr(skb)->hop_limit--;
@@ -623,7 +623,7 @@ static int ipv6_rpl_srh_rcv(struct sk_buff *skb, struct inet6_dev *idev)
__IP6_INC_STATS(net, idev, IPSTATS_MIB_INHDRERRORS);
icmpv6_send(skb, ICMPV6_TIME_EXCEED,
ICMPV6_EXC_HOPLIMIT, 0);
- kfree_skb(skb);
+ kfree_skb_reason(skb, SKB_DROP_REASON_IP_TTL_EXCEEDED);
return -1;
}
ipv6_hdr(skb)->hop_limit--;
@@ -815,7 +815,7 @@ static int ipv6_rthdr_rcv(struct sk_buff *skb)
__IP6_INC_STATS(net, idev, IPSTATS_MIB_INHDRERRORS);
icmpv6_send(skb, ICMPV6_TIME_EXCEED, ICMPV6_EXC_HOPLIMIT,
0);
- kfree_skb(skb);
+ kfree_skb_reason(skb, SKB_DROP_REASON_IP_TTL_EXCEEDED);
return -1;
}
ipv6_hdr(skb)->hop_limit--;
diff --git a/net/ipv6/ip6_output.c b/net/ipv6/ip6_output.c
index 8fc4766c8da9..0b6d78c8b6be 100644
--- a/net/ipv6/ip6_output.c
+++ b/net/ipv6/ip6_output.c
@@ -577,7 +577,7 @@ int ip6_forward(struct sk_buff *skb)
icmpv6_send(skb, ICMPV6_TIME_EXCEED, ICMPV6_EXC_HOPLIMIT, 0);
__IP6_INC_STATS(net, idev, IPSTATS_MIB_INHDRERRORS);
- kfree_skb_reason(skb, SKB_DROP_REASON_IP_INHDR);
+ kfree_skb_reason(skb, SKB_DROP_REASON_IP_TTL_EXCEEDED);
return -ETIMEDOUT;
}
--
2.43.0
on 9/4/26 11:01 AM, Junjie Cao wrote:
> The forwarding paths report an expired TTL or hop limit as
> SKB_DROP_REASON_IP_INHDR, the reason otherwise used for a header that is
> malformed (ip_input.c, exthdrs.c, br_netfilter). Nothing else in the drop
> path separates the two: IPSTATS_MIB_INHDRERRORS covers both, and the TTL
> check runs before NF_INET_FORWARD, so netfilter tracing stops at
> PREROUTING and never sees the drop.
>
> The Fedora bug linked below shows how that reads in practice. The
> reporter took kfree_skb(reason=IP_INHDR, loc=ip_forward) to mean the
> software header checksum check had failed, and worked through RX checksum
> offload, tc csum actions and both libvirt firewall backends before the
> drops turned out to be replies arriving with TTL 1. ip_forward() never
> verifies the header checksum; that runs earlier, in ip_rcv_core(), and
> reports IP_CSUM.
>
> TTL expiry is not a corner case -- every traceroute through a Linux
> router goes through too_many_hops.
>
> The three loopback hop limit checks in exthdrs.c drop with no reason at
> all; give them the new one.
>
> IPSTATS_MIB_INHDRERRORS stays as it is: RFC 1213 counts time-to-live
> exceeded under ipInHdrErrors. The drop reason has no such constraint.
>
> Link: https://bugzilla.redhat.com/show_bug.cgi?id=2517131
> Signed-off-by: Junjie Cao <junjie.cao@intel.com>
Reviewed-by: Jiayuan Chen <jiayuan.chen@linux.dev>
[...]
> /**
> @@ -606,6 +607,11 @@ enum skb_drop_reason {
> SKB_DROP_REASON_PSP_OUTPUT,
> /** @SKB_DROP_REASON_RECURSION_LIMIT: Dead loop on virtual device. */
> SKB_DROP_REASON_RECURSION_LIMIT,
> + /**
> + * @SKB_DROP_REASON_IP_TTL_EXCEEDED: IPv4 TTL or IPv6 hop limit hit
> + * zero (see IPSTATS_MIB_INHDRERRORS)
nit: "<= 1" would be more accurate than "hit zero".
> + */
> + SKB_DROP_REASON_IP_TTL_EXCEEDED,
> /**
> * @SKB_DROP_REASON_MAX: the maximum of core drop reasons, which
> * shouldn't be used as a real 'reason' - only for tracing code gen
>
© 2016 - 2026 Red Hat, Inc.