3 Commits
main ... mss

Author SHA1 Message Date
tuna2134
8d2108a727 fuck 2026-07-11 12:18:03 +09:00
tuna2134
f305045a0b fix 2026-07-11 12:14:36 +09:00
tuna2134
8bd7e23a2e support gso 2026-07-11 12:01:00 +09:00
2 changed files with 42 additions and 131 deletions

View File

@@ -37,12 +37,10 @@ sudo ip link set eip0 up
sudo ip addr add 192.0.2.1/30 dev eip0 sudo ip addr add 192.0.2.1/30 dev eip0
``` ```
TCP SYN に MSS オプションがある場合は、外側 IPv6 経路 MTU とトンネル MTU の
小さい方に収まるよう、送信時に MSS を自動的に縮小します。IPv4/IPv6、
802.1Q/802.1ad VLAN、IPv6 拡張ヘッダーに対応し、既に小さい MSS は変更しません。
## MTU と IPv6 フラグメント ## MTU と IPv6 フラグメント
GSO パケットは送信時にソフトウェアで分割してから EtherIP カプセル化します。
EtherIP over IPv6 では、内側 Ethernet ヘッダー 14 bytes、EtherIP ヘッダー EtherIP over IPv6 では、内側 Ethernet ヘッダー 14 bytes、EtherIP ヘッダー
2 bytes、外側 IPv6 ヘッダー 40 bytes の合計 56 bytes が追加されます。そのため、 2 bytes、外側 IPv6 ヘッダー 40 bytes の合計 56 bytes が追加されます。そのため、
`eip0` と外側インターフェースの MTU がともに 1500 の場合、最大サイズの内側 `eip0` と外側インターフェースの MTU がともに 1500 の場合、最大サイズの内側
@@ -67,8 +65,7 @@ sudo sysctl -w net.ipv6.ip6frag_time=10
この設定は経路上でのフラグメント欠落を防ぐものではありません。フラグメントを この設定は経路上でのフラグメント欠落を防ぐものではありません。フラグメントを
避けるには、外側インターフェースと経路の MTU を 1556 以上にするか、`eip0` 避けるには、外側インターフェースと経路の MTU を 1556 以上にするか、`eip0`
MTU を外側 MTU から 56 引いた値以下(外側 MTU 1500 なら 1444 以下)に設定して MTU を外側 MTU から 56 引いた値以下(外側 MTU 1500 なら 1444 以下)に設定して
ください。TCP は上記の MSS clamping でフラグメントを回避できますが、TCP 以外の ください。
Ethernet フレームには適用されません。
## トンネルの削除 ## トンネルの削除

View File

@@ -2,10 +2,10 @@
#include <linux/etherdevice.h> #include <linux/etherdevice.h>
#include <linux/if_ether.h> #include <linux/if_ether.h>
#include <linux/if_link.h> #include <linux/if_link.h>
#include <linux/if_vlan.h>
#include <linux/in6.h> #include <linux/in6.h>
#include <linux/ip.h> #include <linux/ip.h>
#include <linux/ipv6.h> #include <linux/ipv6.h>
#include <linux/if_vlan.h>
#include <linux/kernel.h> #include <linux/kernel.h>
#include <linux/list.h> #include <linux/list.h>
#include <linux/module.h> #include <linux/module.h>
@@ -16,16 +16,14 @@
#include <linux/rtnetlink.h> #include <linux/rtnetlink.h>
#include <linux/skbuff.h> #include <linux/skbuff.h>
#include <linux/tcp.h> #include <linux/tcp.h>
#include <linux/unaligned.h>
#include <net/checksum.h>
#include <net/ip.h> #include <net/ip.h>
#include <net/ip6_route.h> #include <net/ip6_route.h>
#include <net/ipv6.h> #include <net/ipv6.h>
#include <net/gso.h>
#include <net/net_namespace.h> #include <net/net_namespace.h>
#include <net/netns/generic.h> #include <net/netns/generic.h>
#include <net/protocol.h> #include <net/protocol.h>
#include <net/rtnetlink.h> #include <net/rtnetlink.h>
#include <net/tcp.h>
#include "etherip6_uapi.h" #include "etherip6_uapi.h"
@@ -38,7 +36,6 @@
#define ETHERIP6_HLEN 2 #define ETHERIP6_HLEN 2
#define ETHERIP6_DEFAULT_HOP_LIMIT 64 #define ETHERIP6_DEFAULT_HOP_LIMIT 64
#define ETHERIP6_MAX_MTU 9000 #define ETHERIP6_MAX_MTU 9000
#define ETHERIP6_OUTER_HLEN (sizeof(struct ipv6hdr) + ETHERIP6_HLEN)
struct etherip6_tunnel { struct etherip6_tunnel {
struct list_head list; struct list_head list;
@@ -147,122 +144,6 @@ static struct etherip6_tunnel *etherip6_lookup_rx(struct net *net,
return etherip6_lookup_unique_rx(net, local, remote, iif, false, false); return etherip6_lookup_unique_rx(net, local, remote, iif, false, false);
} }
/*
* Reduce an advertised MSS only when the encapsulated SYN would otherwise
* exceed the smaller of the tunnel MTU and the current outer path MTU.
* skb_header_pointer() keeps the common linear-skb path allocation-free while
* still handling cloned and non-linear packets correctly.
*/
static void etherip6_clamp_tcp_mss(struct sk_buff *skb, unsigned int path_mtu)
{
struct vlan_hdr vlan_buf, *vh;
struct tcphdr tcp_buf, *th;
struct ethhdr eth_buf, *eth;
unsigned int nhoff = ETH_HLEN;
unsigned int thoff, tcp_hlen;
unsigned int inner_mtu, ip_hlen;
unsigned char opt_buf[MAX_TCP_OPTION_SPACE], *opt;
unsigned int optlen;
unsigned int mss_offset;
__be16 proto;
u16 old_mss, new_mss;
u8 nexthdr;
eth = skb_header_pointer(skb, 0, sizeof(eth_buf), &eth_buf);
if (unlikely(!eth))
return;
proto = eth->h_proto;
while (eth_type_vlan(proto)) {
vh = skb_header_pointer(skb, nhoff, sizeof(vlan_buf), &vlan_buf);
if (unlikely(!vh))
return;
proto = vh->h_vlan_encapsulated_proto;
nhoff += sizeof(*vh);
}
if (unlikely(path_mtu <= ETHERIP6_OUTER_HLEN + nhoff))
return;
inner_mtu = min_t(unsigned int, skb->dev->mtu,
path_mtu - ETHERIP6_OUTER_HLEN - nhoff);
if (proto == htons(ETH_P_IP)) {
struct iphdr ip_buf, *iph;
iph = skb_header_pointer(skb, nhoff, sizeof(ip_buf), &ip_buf);
if (!iph || iph->version != 4 || iph->ihl < 5 ||
iph->protocol != IPPROTO_TCP ||
(iph->frag_off & htons(IP_MF | IP_OFFSET)))
return;
thoff = nhoff + iph->ihl * 4;
ip_hlen = sizeof(struct iphdr);
} else if (proto == htons(ETH_P_IPV6)) {
struct ipv6hdr ip6_buf, *ip6h;
__be16 frag_off = 0;
int offset;
ip6h = skb_header_pointer(skb, nhoff, sizeof(ip6_buf), &ip6_buf);
if (!ip6h || ip6h->version != 6)
return;
nexthdr = ip6h->nexthdr;
offset = ipv6_skip_exthdr(skb, nhoff + sizeof(*ip6h),
&nexthdr, &frag_off);
if (offset < 0 || nexthdr != IPPROTO_TCP || frag_off)
return;
thoff = offset;
ip_hlen = sizeof(struct ipv6hdr);
} else {
return;
}
th = skb_header_pointer(skb, thoff, sizeof(tcp_buf), &tcp_buf);
if (!th || !th->syn || th->doff < sizeof(*th) / 4)
return;
tcp_hlen = th->doff * 4;
if (tcp_hlen > MAX_TCP_HEADER ||
inner_mtu <= ip_hlen + sizeof(struct tcphdr))
return;
new_mss = min_t(unsigned int, U16_MAX,
inner_mtu - ip_hlen - sizeof(struct tcphdr));
optlen = tcp_hlen - sizeof(*th);
opt = skb_header_pointer(skb, thoff + sizeof(*th), optlen, opt_buf);
if (!opt)
return;
while (optlen) {
u8 kind = opt[0];
u8 len;
if (kind == TCPOPT_EOL)
return;
if (kind == TCPOPT_NOP) {
opt++;
optlen--;
continue;
}
if (optlen < 2 || (len = opt[1]) < 2 || len > optlen)
return;
if (kind == TCPOPT_MSS && len == TCPOLEN_MSS) {
old_mss = get_unaligned_be16(opt + 2);
if (old_mss > new_mss) {
mss_offset = thoff + tcp_hlen - optlen + 2;
if (skb_ensure_writable(skb, mss_offset + 2))
return;
th = (struct tcphdr *)(skb->data + thoff);
put_unaligned_be16(new_mss,
skb->data + mss_offset);
inet_proto_csum_replace2(&th->check, skb,
htons(old_mss), htons(new_mss),
false);
}
return;
}
opt += len;
optlen -= len;
}
}
static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev) static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
{ {
struct etherip6_tunnel *tun = netdev_priv(dev); struct etherip6_tunnel *tun = netdev_priv(dev);
@@ -276,6 +157,40 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
unsigned int inner_len; unsigned int inner_len;
int headroom; int headroom;
int err; int err;
struct sk_buff *segs, *next;
unsigned int gso_limit;
if (skb_is_gso(skb)) {
/*
* GSO is performed before EtherIP/IPv6 encapsulation. Limit the
* TCP payload according to the configured inner MTU; this preserves
* both 1500-byte and jumbo-frame tunnel configurations.
*/
gso_limit = max_t(unsigned int, dev->mtu,
ETH_HLEN + ETHERIP6_HLEN + sizeof(struct ipv6hdr) +
sizeof(struct tcphdr)) - ETH_HLEN - ETHERIP6_HLEN -
(skb->protocol == htons(ETH_P_IPV6) ?
sizeof(struct ipv6hdr) : sizeof(struct iphdr)) -
sizeof(struct tcphdr);
if (skb_shinfo(skb)->gso_size > gso_limit)
skb_shinfo(skb)->gso_size = gso_limit;
segs = skb_gso_segment(skb, 0);
if (IS_ERR(segs))
goto tx_error;
if (!segs) {
dev_kfree_skb(skb);
return NETDEV_TX_OK;
}
consume_skb(skb);
while (segs) {
next = segs->next;
segs->next = NULL;
etherip6_xmit(segs, dev);
segs = next;
}
return NETDEV_TX_OK;
}
if (skb->len + ETHERIP6_HLEN > IPV6_MAXPLEN) if (skb->len + ETHERIP6_HLEN > IPV6_MAXPLEN)
goto tx_error; goto tx_error;
@@ -291,8 +206,6 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
goto tx_error; goto tx_error;
} }
etherip6_clamp_tcp_mss(skb, dst_mtu(dst));
headroom = LL_RESERVED_SPACE(dst->dev) + sizeof(*ip6h) + ETHERIP6_HLEN; headroom = LL_RESERVED_SPACE(dst->dev) + sizeof(*ip6h) + ETHERIP6_HLEN;
err = skb_cow_head(skb, headroom); err = skb_cow_head(skb, headroom);
if (err) { if (err) {
@@ -389,8 +302,9 @@ static void etherip6_setup(struct net_device *dev)
dev->needs_free_netdev = true; dev->needs_free_netdev = true;
dev->type = ARPHRD_ETHER; dev->type = ARPHRD_ETHER;
dev->flags &= ~IFF_NOARP; dev->flags &= ~IFF_NOARP;
dev->features &= ~(NETIF_F_GSO_MASK | NETIF_F_CSUM_MASK); dev->features &= ~NETIF_F_CSUM_MASK;
dev->hw_features = 0; dev->features |= NETIF_F_GSO_SOFTWARE;
dev->hw_features = NETIF_F_GSO_SOFTWARE;
dev->vlan_features = 0; dev->vlan_features = 0;
dev->min_mtu = ETH_MIN_MTU; dev->min_mtu = ETH_MIN_MTU;
dev->max_mtu = ETHERIP6_MAX_MTU; dev->max_mtu = ETHERIP6_MAX_MTU;