diff --git a/Makefile b/Makefile index 5552778..74214d7 100644 --- a/Makefile +++ b/Makefile @@ -10,11 +10,6 @@ all: module tools module: $(MAKE) -C $(KDIR) M=$(PWD) modules -tools: etherip6ctl - -etherip6ctl: etherip6ctl.c etherip6_uapi.h - $(CC) $(CFLAGS) -Wall -Wextra -O2 -o $@ etherip6ctl.c - clean: @if [ -d "$(KDIR)" ]; then \ $(MAKE) -C "$(KDIR)" M="$(PWD)" clean; \ diff --git a/README.md b/README.md index 699e0e6..5e63564 100644 --- a/README.md +++ b/README.md @@ -37,8 +37,41 @@ sudo ip link set eip0 up sudo ip addr add 192.0.2.1/30 dev eip0 ``` +TCP SYN に MSS オプションがある場合、外側 IPv6 経路の MTU を超えないよう、 +送信時に IPv4/IPv6 の MSS を自動的に縮小します。802.1Q/802.1ad VLAN と IPv6 +拡張ヘッダーにも対応します。既に十分小さい MSS は変更しません。 + +## MTU と IPv6 フラグメント + +EtherIP over IPv6 では、内側 Ethernet ヘッダー 14 bytes、EtherIP ヘッダー +2 bytes、外側 IPv6 ヘッダー 40 bytes の合計 56 bytes が追加されます。そのため、 +`eip0` と外側インターフェースの MTU がともに 1500 の場合、最大サイズの内側 +フレームは外側で 1556 bytes となり、IPv6 フラグメントが必ず発生します。 + +フラグメントの一部が欠落した場合、受信側ではパケット全体を再構成できません。 +再構成の状態は次のコマンドで確認できます。 + +```sh +nstat -az | grep Ip6Reasm +``` + +`Ip6ReasmFails` または `Ip6ReasmTimeout` が転送中に増える場合は、フラグメントの +欠落、または受信側の再構成キュー不足が発生しています。再構成キュー不足に対する +緩和策として、受信側で次の値を設定できます。 + +```sh +sudo sysctl -w net.ipv6.ip6frag_high_thresh=268435456 +sudo sysctl -w net.ipv6.ip6frag_time=10 +``` + +この設定は経路上でのフラグメント欠落を防ぐものではありません。フラグメントを +避けるには、外側インターフェースと経路の MTU を 1556 以上にするか、`eip0` の +MTU を外側 MTU から 56 引いた値以下(外側 MTU 1500 なら 1444 以下)に設定して +ください。TCP については MSS の自動縮小で回避できる場合がありますが、任意の +Ethernet フレームには適用できません。 + ## トンネルの削除 ```sh sudo etherip-client delete -d eip0 -``` \ No newline at end of file +``` diff --git a/etherip6.c b/etherip6.c index f3d4d24..4ed5fc4 100644 --- a/etherip6.c +++ b/etherip6.c @@ -5,6 +5,7 @@ #include #include #include +#include #include #include #include @@ -14,12 +15,17 @@ #include #include #include +#include +#include +#include +#include #include #include #include #include #include #include +#include #include "etherip6_uapi.h" @@ -33,6 +39,9 @@ #define ETHERIP6_DEFAULT_HOP_LIMIT 64 #define ETHERIP6_MAX_MTU 9000 +/* IPv6 + EtherIP headers added outside the encapsulated Ethernet frame. */ +#define ETHERIP6_OUTER_HLEN (sizeof(struct ipv6hdr) + ETHERIP6_HLEN) + struct etherip6_tunnel { struct list_head list; struct net_device *dev; @@ -140,6 +149,110 @@ static struct etherip6_tunnel *etherip6_lookup_rx(struct net *net, return etherip6_lookup_unique_rx(net, local, remote, iif, false, false); } +static void etherip6_clamp_tcp_mss(struct sk_buff *skb, unsigned int path_mtu) +{ + struct vlan_hdr _vh, *vh; + struct tcphdr _th, *th; + struct ethhdr _eth, *eth; + unsigned int nhoff = ETH_HLEN; + unsigned int thoff, tcp_hlen; + unsigned int inner_mtu; + unsigned int min_ip_hlen; + unsigned char *opt; + unsigned int optlen; + __be16 proto; + u16 old_mss, new_mss; + u8 nexthdr; + + eth = skb_header_pointer(skb, 0, sizeof(_eth), &_eth); + if (!eth) + return; + proto = eth->h_proto; + + while (eth_type_vlan(proto)) { + vh = skb_header_pointer(skb, nhoff, sizeof(_vh), &_vh); + if (!vh) + return; + proto = vh->h_vlan_encapsulated_proto; + nhoff += sizeof(*vh); + } + + if (path_mtu <= ETHERIP6_OUTER_HLEN + nhoff) + return; + inner_mtu = min_t(unsigned int, skb->dev->mtu, + path_mtu - ETHERIP6_OUTER_HLEN - nhoff); + + if (proto == htons(ETH_P_IP)) { + struct iphdr _iph, *iph; + + iph = skb_header_pointer(skb, nhoff, sizeof(_iph), &_iph); + if (!iph || iph->version != 4 || iph->ihl < 5 || + iph->protocol != IPPROTO_TCP || + (iph->frag_off & htons(IP_MF | IP_OFFSET))) + return; + thoff = nhoff + iph->ihl * 4; + min_ip_hlen = sizeof(struct iphdr); + } else if (proto == htons(ETH_P_IPV6)) { + struct ipv6hdr _ip6h, *ip6h; + __be16 frag_off = 0; + int offset; + + ip6h = skb_header_pointer(skb, nhoff, sizeof(_ip6h), &_ip6h); + if (!ip6h || ip6h->version != 6) + return; + nexthdr = ip6h->nexthdr; + offset = ipv6_skip_exthdr(skb, nhoff + sizeof(*ip6h), + &nexthdr, &frag_off); + if (offset < 0 || nexthdr != IPPROTO_TCP || frag_off) + return; + thoff = offset; + min_ip_hlen = sizeof(struct ipv6hdr); + } else { + return; + } + + th = skb_header_pointer(skb, thoff, sizeof(_th), &_th); + if (!th || !th->syn || th->doff < sizeof(*th) / 4) + return; + tcp_hlen = th->doff * 4; + if (inner_mtu <= min_ip_hlen + sizeof(*th)) + return; + new_mss = min_t(unsigned int, U16_MAX, + inner_mtu - min_ip_hlen - sizeof(*th)); + + if (skb_ensure_writable(skb, thoff + tcp_hlen)) + return; + th = (struct tcphdr *)(skb->data + thoff); + opt = (unsigned char *)(th + 1); + optlen = tcp_hlen - sizeof(*th); + + while (optlen) { + u8 kind = opt[0]; + u8 len; + + if (kind == TCPOPT_EOL) + return; + if (kind == TCPOPT_NOP) { + opt++; + optlen--; + continue; + } + if (optlen < 2 || (len = opt[1]) < 2 || len > optlen) + return; + if (kind == TCPOPT_MSS && len == TCPOLEN_MSS) { + old_mss = get_unaligned_be16(opt + 2); + if (old_mss > new_mss) { + put_unaligned_be16(new_mss, opt + 2); + inet_proto_csum_replace2(&th->check, skb, + htons(old_mss), htons(new_mss), false); + } + return; + } + opt += len; + optlen -= len; + } +} + static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev) { struct etherip6_tunnel *tun = netdev_priv(dev); @@ -150,6 +263,7 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev) struct pcpu_sw_netstats *stats; __be16 *etherip; unsigned int payload_len; + unsigned int inner_len; int headroom; int err; @@ -167,6 +281,8 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev) goto tx_error; } + etherip6_clamp_tcp_mss(skb, dst_mtu(dst)); + headroom = LL_RESERVED_SPACE(dst->dev) + sizeof(*ip6h) + ETHERIP6_HLEN; err = skb_cow_head(skb, headroom); if (err) { @@ -175,6 +291,7 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev) } payload_len = skb->len + ETHERIP6_HLEN; + inner_len = skb->len; etherip = skb_push(skb, ETHERIP6_HLEN); *etherip = ETHERIP6_HDR; @@ -194,13 +311,28 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev) skb->ignore_df = 1; skb_dst_set(skb, dst); + /* + * skb->cb belongs to the protocol currently processing the skb. The + * encapsulated frame may leave bridge, qdisc, or inner IPv6 state in it; + * in particular, stale inet6_skb_parm flags or frag_max_size corrupt the + * outer IPv6 fragmentation path. Native IPv6 tunnels clear this state + * in ip6tunnel_xmit() before calling ip6_local_out(). + */ + memset(skb->cb, 0, sizeof(struct inet6_skb_parm)); + + err = ip6_local_out(net, NULL, skb); + if (unlikely(net_xmit_eval(err))) { + atomic_long_inc(&tun->tx_dropped); + return NETDEV_TX_OK; + } + stats = this_cpu_ptr(tun->stats); u64_stats_update_begin(&stats->syncp); u64_stats_inc(&stats->tx_packets); - u64_stats_add(&stats->tx_bytes, payload_len - ETHERIP6_HLEN); + u64_stats_add(&stats->tx_bytes, inner_len); u64_stats_update_end(&stats->syncp); - return ip6_local_out(net, NULL, skb) == 0 ? NETDEV_TX_OK : NETDEV_TX_OK; + return NETDEV_TX_OK; tx_error: atomic_long_inc(&tun->tx_dropped);