Compare commits
5 Commits
bbf52550be
...
patch
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fccfaf6249 | ||
| 5ac8f37aca | |||
|
|
1333e033bb | ||
|
|
b6240ba235 | ||
|
|
5fa3637a5b |
39
README.md
39
README.md
@@ -37,9 +37,42 @@ sudo ip link set eip0 up
|
||||
sudo ip addr add 192.0.2.1/30 dev eip0
|
||||
```
|
||||
|
||||
TCP SYN に MSS オプションがある場合、外側 IPv6 経路の MTU を超えないよう、
|
||||
送信時に IPv4/IPv6 の MSS を自動的に縮小します。802.1Q/802.1ad VLAN と IPv6
|
||||
拡張ヘッダーにも対応します。既に十分小さい MSS は変更しません。
|
||||
`eip0` は TCP/UDP/SCTP の software GSO に対応します。GSO パケットは内側の
|
||||
Ethernet フレームとしてセグメント化してから、それぞれを EtherIP でカプセル化
|
||||
します。
|
||||
|
||||
## MTU と IPv6 フラグメント
|
||||
|
||||
EtherIP over IPv6 では、内側 Ethernet ヘッダー 14 bytes、EtherIP ヘッダー
|
||||
2 bytes、外側 IPv6 ヘッダー 40 bytes の合計 56 bytes が追加されます。そのため、
|
||||
`eip0` と外側インターフェースの MTU がともに 1500 の場合、最大サイズの内側
|
||||
フレームは外側で 1556 bytes となり、IPv6 フラグメントが必ず発生します。
|
||||
|
||||
フラグメントの一部が欠落した場合、受信側ではパケット全体を再構成できません。
|
||||
再構成の状態は次のコマンドで確認できます。
|
||||
|
||||
```sh
|
||||
nstat -az | grep Ip6Reasm
|
||||
```
|
||||
|
||||
`Ip6ReasmFails` または `Ip6ReasmTimeout` が転送中に増える場合は、フラグメントの
|
||||
欠落、または受信側の再構成キュー不足が発生しています。再構成キュー不足に対する
|
||||
緩和策として、受信側で次の値を設定できます。
|
||||
|
||||
```sh
|
||||
sudo sysctl -w net.ipv6.ip6frag_high_thresh=268435456
|
||||
sudo sysctl -w net.ipv6.ip6frag_time=10
|
||||
```
|
||||
|
||||
この設定は経路上でのフラグメント欠落を防ぐものではありません。フラグメントを
|
||||
避けるには、外側インターフェースと経路の MTU を 1556 以上にするか、`eip0` の
|
||||
MTU を外側 MTU から 56 引いた値以下(外側 MTU 1500 なら 1444 以下)に設定して
|
||||
ください。
|
||||
|
||||
一方、`eip0` の MTU は最大 9000 に設定できます。外側経路の MTU より大きい
|
||||
非 GSO パケット、および GSO の各セグメントは、外側 IPv6 でフラグメント化して
|
||||
送信されます。したがって MTU 9000 のフレームも転送できますが、経路上で IPv6
|
||||
フラグメントが欠落しないことと、受信側に十分な再構成キューが必要です。
|
||||
|
||||
## トンネルの削除
|
||||
|
||||
|
||||
175
etherip6.c
175
etherip6.c
@@ -3,9 +3,7 @@
|
||||
#include <linux/if_ether.h>
|
||||
#include <linux/if_link.h>
|
||||
#include <linux/in6.h>
|
||||
#include <linux/ip.h>
|
||||
#include <linux/ipv6.h>
|
||||
#include <linux/if_vlan.h>
|
||||
#include <linux/kernel.h>
|
||||
#include <linux/list.h>
|
||||
#include <linux/module.h>
|
||||
@@ -15,17 +13,13 @@
|
||||
#include <linux/rculist.h>
|
||||
#include <linux/rtnetlink.h>
|
||||
#include <linux/skbuff.h>
|
||||
#include <linux/tcp.h>
|
||||
#include <linux/unaligned.h>
|
||||
#include <net/checksum.h>
|
||||
#include <net/ip.h>
|
||||
#include <net/ip6_route.h>
|
||||
#include <net/ipv6.h>
|
||||
#include <net/net_namespace.h>
|
||||
#include <net/netns/generic.h>
|
||||
#include <net/gso.h>
|
||||
#include <net/protocol.h>
|
||||
#include <net/rtnetlink.h>
|
||||
#include <net/tcp.h>
|
||||
|
||||
#include "etherip6_uapi.h"
|
||||
|
||||
@@ -38,9 +32,7 @@
|
||||
#define ETHERIP6_HLEN 2
|
||||
#define ETHERIP6_DEFAULT_HOP_LIMIT 64
|
||||
#define ETHERIP6_MAX_MTU 9000
|
||||
|
||||
/* IPv6 + EtherIP headers added outside the encapsulated Ethernet frame. */
|
||||
#define ETHERIP6_OUTER_HLEN (sizeof(struct ipv6hdr) + ETHERIP6_HLEN)
|
||||
#define ETHERIP6_NEEDED_HEADROOM (sizeof(struct ipv6hdr) + ETHERIP6_HLEN)
|
||||
|
||||
struct etherip6_tunnel {
|
||||
struct list_head list;
|
||||
@@ -149,111 +141,7 @@ static struct etherip6_tunnel *etherip6_lookup_rx(struct net *net,
|
||||
return etherip6_lookup_unique_rx(net, local, remote, iif, false, false);
|
||||
}
|
||||
|
||||
static void etherip6_clamp_tcp_mss(struct sk_buff *skb, unsigned int path_mtu)
|
||||
{
|
||||
struct vlan_hdr _vh, *vh;
|
||||
struct tcphdr _th, *th;
|
||||
struct ethhdr _eth, *eth;
|
||||
unsigned int nhoff = ETH_HLEN;
|
||||
unsigned int thoff, tcp_hlen;
|
||||
unsigned int inner_mtu;
|
||||
unsigned int min_ip_hlen;
|
||||
unsigned char *opt;
|
||||
unsigned int optlen;
|
||||
__be16 proto;
|
||||
u16 old_mss, new_mss;
|
||||
u8 nexthdr;
|
||||
|
||||
eth = skb_header_pointer(skb, 0, sizeof(_eth), &_eth);
|
||||
if (!eth)
|
||||
return;
|
||||
proto = eth->h_proto;
|
||||
|
||||
while (eth_type_vlan(proto)) {
|
||||
vh = skb_header_pointer(skb, nhoff, sizeof(_vh), &_vh);
|
||||
if (!vh)
|
||||
return;
|
||||
proto = vh->h_vlan_encapsulated_proto;
|
||||
nhoff += sizeof(*vh);
|
||||
}
|
||||
|
||||
if (path_mtu <= ETHERIP6_OUTER_HLEN + nhoff)
|
||||
return;
|
||||
inner_mtu = min_t(unsigned int, skb->dev->mtu,
|
||||
path_mtu - ETHERIP6_OUTER_HLEN - nhoff);
|
||||
|
||||
if (proto == htons(ETH_P_IP)) {
|
||||
struct iphdr _iph, *iph;
|
||||
|
||||
iph = skb_header_pointer(skb, nhoff, sizeof(_iph), &_iph);
|
||||
if (!iph || iph->version != 4 || iph->ihl < 5 ||
|
||||
iph->protocol != IPPROTO_TCP ||
|
||||
(iph->frag_off & htons(IP_MF | IP_OFFSET)))
|
||||
return;
|
||||
thoff = nhoff + iph->ihl * 4;
|
||||
min_ip_hlen = sizeof(struct iphdr);
|
||||
} else if (proto == htons(ETH_P_IPV6)) {
|
||||
struct ipv6hdr _ip6h, *ip6h;
|
||||
__be16 frag_off = 0;
|
||||
int offset;
|
||||
|
||||
ip6h = skb_header_pointer(skb, nhoff, sizeof(_ip6h), &_ip6h);
|
||||
if (!ip6h || ip6h->version != 6)
|
||||
return;
|
||||
nexthdr = ip6h->nexthdr;
|
||||
offset = ipv6_skip_exthdr(skb, nhoff + sizeof(*ip6h),
|
||||
&nexthdr, &frag_off);
|
||||
if (offset < 0 || nexthdr != IPPROTO_TCP || frag_off)
|
||||
return;
|
||||
thoff = offset;
|
||||
min_ip_hlen = sizeof(struct ipv6hdr);
|
||||
} else {
|
||||
return;
|
||||
}
|
||||
|
||||
th = skb_header_pointer(skb, thoff, sizeof(_th), &_th);
|
||||
if (!th || !th->syn || th->doff < sizeof(*th) / 4)
|
||||
return;
|
||||
tcp_hlen = th->doff * 4;
|
||||
if (inner_mtu <= min_ip_hlen + sizeof(*th))
|
||||
return;
|
||||
new_mss = min_t(unsigned int, U16_MAX,
|
||||
inner_mtu - min_ip_hlen - sizeof(*th));
|
||||
|
||||
if (skb_ensure_writable(skb, thoff + tcp_hlen))
|
||||
return;
|
||||
th = (struct tcphdr *)(skb->data + thoff);
|
||||
opt = (unsigned char *)(th + 1);
|
||||
optlen = tcp_hlen - sizeof(*th);
|
||||
|
||||
while (optlen) {
|
||||
u8 kind = opt[0];
|
||||
u8 len;
|
||||
|
||||
if (kind == TCPOPT_EOL)
|
||||
return;
|
||||
if (kind == TCPOPT_NOP) {
|
||||
opt++;
|
||||
optlen--;
|
||||
continue;
|
||||
}
|
||||
if (optlen < 2 || (len = opt[1]) < 2 || len > optlen)
|
||||
return;
|
||||
if (kind == TCPOPT_MSS && len == TCPOLEN_MSS) {
|
||||
old_mss = get_unaligned_be16(opt + 2);
|
||||
if (old_mss > new_mss) {
|
||||
put_unaligned_be16(new_mss, opt + 2);
|
||||
inet_proto_csum_replace2(&th->check, skb,
|
||||
htons(old_mss), htons(new_mss), false);
|
||||
}
|
||||
return;
|
||||
}
|
||||
opt += len;
|
||||
optlen -= len;
|
||||
}
|
||||
}
|
||||
|
||||
static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
|
||||
static void etherip6_xmit_one(struct sk_buff *skb, struct net_device *dev)
|
||||
{
|
||||
struct etherip6_tunnel *tun = netdev_priv(dev);
|
||||
struct net *net = dev_net(dev);
|
||||
@@ -263,6 +151,7 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
|
||||
struct pcpu_sw_netstats *stats;
|
||||
__be16 *etherip;
|
||||
unsigned int payload_len;
|
||||
unsigned int inner_len;
|
||||
int headroom;
|
||||
int err;
|
||||
|
||||
@@ -280,8 +169,6 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
|
||||
goto tx_error;
|
||||
}
|
||||
|
||||
etherip6_clamp_tcp_mss(skb, dst_mtu(dst));
|
||||
|
||||
headroom = LL_RESERVED_SPACE(dst->dev) + sizeof(*ip6h) + ETHERIP6_HLEN;
|
||||
err = skb_cow_head(skb, headroom);
|
||||
if (err) {
|
||||
@@ -290,6 +177,7 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
|
||||
}
|
||||
|
||||
payload_len = skb->len + ETHERIP6_HLEN;
|
||||
inner_len = skb->len;
|
||||
|
||||
etherip = skb_push(skb, ETHERIP6_HLEN);
|
||||
*etherip = ETHERIP6_HDR;
|
||||
@@ -318,17 +206,57 @@ static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
|
||||
*/
|
||||
memset(skb->cb, 0, sizeof(struct inet6_skb_parm));
|
||||
|
||||
err = ip6_local_out(net, NULL, skb);
|
||||
if (unlikely(net_xmit_eval(err))) {
|
||||
atomic_long_inc(&tun->tx_dropped);
|
||||
return;
|
||||
}
|
||||
|
||||
stats = this_cpu_ptr(tun->stats);
|
||||
u64_stats_update_begin(&stats->syncp);
|
||||
u64_stats_inc(&stats->tx_packets);
|
||||
u64_stats_add(&stats->tx_bytes, payload_len - ETHERIP6_HLEN);
|
||||
u64_stats_add(&stats->tx_bytes, inner_len);
|
||||
u64_stats_update_end(&stats->syncp);
|
||||
|
||||
return ip6_local_out(net, NULL, skb) == 0 ? NETDEV_TX_OK : NETDEV_TX_OK;
|
||||
return;
|
||||
|
||||
tx_error:
|
||||
atomic_long_inc(&tun->tx_dropped);
|
||||
dev_kfree_skb(skb);
|
||||
return;
|
||||
}
|
||||
|
||||
static netdev_tx_t etherip6_xmit(struct sk_buff *skb, struct net_device *dev)
|
||||
{
|
||||
struct etherip6_tunnel *tun = netdev_priv(dev);
|
||||
struct sk_buff *segs, *nskb;
|
||||
|
||||
if (!skb_is_gso(skb)) {
|
||||
etherip6_xmit_one(skb, dev);
|
||||
return NETDEV_TX_OK;
|
||||
}
|
||||
|
||||
/*
|
||||
* EtherIP has no hardware GSO type. Segment the inner Ethernet frame
|
||||
* before adding the EtherIP and outer IPv6 headers. Each resulting skb
|
||||
* is then a normal IPv6 packet and can be fragmented by ip6_local_out()
|
||||
* when the outer path MTU is smaller than the tunnel MTU.
|
||||
*/
|
||||
segs = skb_gso_segment(skb, 0);
|
||||
if (IS_ERR_OR_NULL(segs)) {
|
||||
atomic_long_inc(&tun->tx_dropped);
|
||||
dev_kfree_skb(skb);
|
||||
return NETDEV_TX_OK;
|
||||
}
|
||||
|
||||
consume_skb(skb);
|
||||
while (segs) {
|
||||
nskb = segs;
|
||||
segs = segs->next;
|
||||
nskb->next = NULL;
|
||||
etherip6_xmit_one(nskb, dev);
|
||||
}
|
||||
|
||||
return NETDEV_TX_OK;
|
||||
}
|
||||
|
||||
@@ -371,9 +299,12 @@ static void etherip6_setup(struct net_device *dev)
|
||||
dev->needs_free_netdev = true;
|
||||
dev->type = ARPHRD_ETHER;
|
||||
dev->flags &= ~IFF_NOARP;
|
||||
dev->features &= ~(NETIF_F_GSO_MASK | NETIF_F_CSUM_MASK);
|
||||
dev->hw_features = 0;
|
||||
dev->vlan_features = 0;
|
||||
dev->features |= NETIF_F_SG | NETIF_F_HW_CSUM |
|
||||
NETIF_F_GSO_SOFTWARE;
|
||||
dev->hw_features |= NETIF_F_SG | NETIF_F_HW_CSUM |
|
||||
NETIF_F_GSO_SOFTWARE;
|
||||
dev->vlan_features = dev->features;
|
||||
dev->needed_headroom = ETHERIP6_NEEDED_HEADROOM;
|
||||
dev->min_mtu = ETH_MIN_MTU;
|
||||
dev->max_mtu = ETHERIP6_MAX_MTU;
|
||||
eth_hw_addr_random(dev);
|
||||
|
||||
Reference in New Issue
Block a user