ipv4: tcp: get rid of ugly unicast_sock

author Eric Dumazet <edumazet@google.com>

Fri, 30 Jan 2015 05:35:05 +0000 (21:35 -0800)

committer Luis Henriques <luis.henriques@canonical.com>

Tue, 10 Feb 2015 13:38:50 +0000 (13:38 +0000)
author Eric Dumazet <edumazet@google.com>
Fri, 30 Jan 2015 05:35:05 +0000 (21:35 -0800)
committer Luis Henriques <luis.henriques@canonical.com>
Tue, 10 Feb 2015 13:38:50 +0000 (13:38 +0000)
diff --git a/include/net/ip.h b/include/net/ip.h

index 3e8f0b32d31ba3897daa3d124ba516a89e5b5cbb..fdef22d8820334397d364b650179732403a5dc12 100644 (file)
--- a/include/net/ip.h
+++ b/include/net/ip.h
@@ -180,7 +180,7 @@ static inline __u8 ip_reply_arg_flowi_flags(const struct ip_reply_arg *arg)
         return (arg->flags & IP_REPLY_ARG_NOSRCCHECK) ? FLOWI_FLAG_ANYSRC : 0;
  }
  
-void ip_send_unicast_reply(struct net *net, struct sk_buff *skb, __be32 daddr,
+void ip_send_unicast_reply(struct sock *sk, struct sk_buff *skb, __be32 daddr,
                            __be32 saddr, const struct ip_reply_arg *arg,
                            unsigned int len);
  
diff --git a/include/net/netns/ipv4.h b/include/net/netns/ipv4.h

index aec5e12f9f19f1a6c506e47f60cc3056d7ce2a3d..80a1c572b9a0d368151a0c6d3774902f2904bcbb 100644 (file)
--- a/include/net/netns/ipv4.h
+++ b/include/net/netns/ipv4.h
@@ -52,6 +52,7 @@ struct netns_ipv4 {
         struct inet_peer_base   *peers;
         struct tcpm_hash_bucket *tcp_metrics_hash;
         unsigned int            tcp_metrics_hash_log;
+       struct sock  * __percpu *tcp_sk;
         struct netns_frags      frags;
  #ifdef CONFIG_NETFILTER
         struct xt_table         *iptable_filter;
diff --git a/net/ipv4/ip_output.c b/net/ipv4/ip_output.c

index 34ee97c6aced0cb53abf74c0bc487d6da0a2c6c8..4aca72f52636e97b645597e9b77344e93c8f40b0 100644 (file)
--- a/net/ipv4/ip_output.c
+++ b/net/ipv4/ip_output.c
@@ -1501,24 +1501,8 @@ static int ip_reply_glue_bits(void *dptr, char *to, int offset,
  /*
   *     Generic function to send a packet as reply to another packet.
   *     Used to send some TCP resets/acks so far.
- *
- *     Use a fake percpu inet socket to avoid false sharing and contention.
   */
-static DEFINE_PER_CPU(struct inet_sock, unicast_sock) = {
-       .sk = {
-               .__sk_common = {
-                       .skc_refcnt = ATOMIC_INIT(1),
-               },
-               .sk_wmem_alloc  = ATOMIC_INIT(1),
-               .sk_allocation  = GFP_ATOMIC,
-               .sk_flags       = (1UL << SOCK_USE_WRITE_QUEUE),
-               .sk_pacing_rate = ~0U,
-       },
-       .pmtudisc       = IP_PMTUDISC_WANT,
-       .uc_ttl         = -1,
-};
-
-void ip_send_unicast_reply(struct net *net, struct sk_buff *skb, __be32 daddr,
+void ip_send_unicast_reply(struct sock *sk, struct sk_buff *skb, __be32 daddr,
                            __be32 saddr, const struct ip_reply_arg *arg,
                            unsigned int len)
  {
@@ -1526,9 +1510,8 @@ void ip_send_unicast_reply(struct net *net, struct sk_buff *skb, __be32 daddr,
         struct ipcm_cookie ipc;
         struct flowi4 fl4;
         struct rtable *rt = skb_rtable(skb);
+       struct net *net = sock_net(sk);
         struct sk_buff *nskb;
-       struct sock *sk;
-       struct inet_sock *inet;
         int err;
  
         if (ip_options_echo(&replyopts.opt.opt, skb))
@@ -1559,15 +1542,11 @@ void ip_send_unicast_reply(struct net *net, struct sk_buff *skb, __be32 daddr,
         if (IS_ERR(rt))
                 return;
  
-       inet = &get_cpu_var(unicast_sock);
+       inet_sk(sk)->tos = arg->tos;
  
-       inet->tos = arg->tos;
-       sk = &inet->sk;
         sk->sk_priority = skb->priority;
         sk->sk_protocol = ip_hdr(skb)->protocol;
         sk->sk_bound_dev_if = arg->bound_dev_if;
-       sock_net_set(sk, net);
-       __skb_queue_head_init(&sk->sk_write_queue);
         sk->sk_sndbuf = sysctl_wmem_default;
         err = ip_append_data(sk, &fl4, ip_reply_glue_bits, arg->iov->iov_base,
                              len, 0, &ipc, &rt, MSG_DONTWAIT);
@@ -1583,13 +1562,10 @@ void ip_send_unicast_reply(struct net *net, struct sk_buff *skb, __be32 daddr,
                           arg->csumoffset) = csum_fold(csum_add(nskb->csum,
                                                                 arg->csum));
                 nskb->ip_summed = CHECKSUM_NONE;
-               skb_orphan(nskb);
                 skb_set_queue_mapping(nskb, skb_get_queue_mapping(skb));
                 ip_push_pending_frames(sk, &fl4);
         }
  out:
-       put_cpu_var(unicast_sock);
-
         ip_rt_put(rt);
  }
  
diff --git a/net/ipv4/tcp_ipv4.c b/net/ipv4/tcp_ipv4.c

index f63c524de5d96497c7e6fd376db178656fb5f7ab..ac7752f4f7771d710b696fd021cc6e96ed36f2b8 100644 (file)
--- a/net/ipv4/tcp_ipv4.c
+++ b/net/ipv4/tcp_ipv4.c
@@ -684,7 +684,8 @@ static void tcp_v4_send_reset(struct sock *sk, struct sk_buff *skb)
  
         net = dev_net(skb_dst(skb)->dev);
         arg.tos = ip_hdr(skb)->tos;
-       ip_send_unicast_reply(net, skb, ip_hdr(skb)->saddr,
+       ip_send_unicast_reply(*this_cpu_ptr(net->ipv4.tcp_sk),
+                             skb, ip_hdr(skb)->saddr,
                               ip_hdr(skb)->daddr, &arg, arg.iov[0].iov_len);
  
         TCP_INC_STATS_BH(net, TCP_MIB_OUTSEGS);
@@ -767,7 +768,8 @@ static void tcp_v4_send_ack(struct sk_buff *skb, u32 seq, u32 ack,
         if (oif)
                 arg.bound_dev_if = oif;
         arg.tos = tos;
-       ip_send_unicast_reply(net, skb, ip_hdr(skb)->saddr,
+       ip_send_unicast_reply(*this_cpu_ptr(net->ipv4.tcp_sk),
+                             skb, ip_hdr(skb)->saddr,
                               ip_hdr(skb)->daddr, &arg, arg.iov[0].iov_len);
  
         TCP_INC_STATS_BH(net, TCP_MIB_OUTSEGS);
@@ -2532,14 +2534,39 @@ struct proto tcp_prot = {
  };
  EXPORT_SYMBOL(tcp_prot);
  
+static void __net_exit tcp_sk_exit(struct net *net)
+{
+       int cpu;
+
+       for_each_possible_cpu(cpu)
+               inet_ctl_sock_destroy(*per_cpu_ptr(net->ipv4.tcp_sk, cpu));
+       free_percpu(net->ipv4.tcp_sk);
+}
+
  static int __net_init tcp_sk_init(struct net *net)
  {
+       int res, cpu;
+
+       net->ipv4.tcp_sk = alloc_percpu(struct sock *);
+       if (!net->ipv4.tcp_sk)
+               return -ENOMEM;
+
+       for_each_possible_cpu(cpu) {
+               struct sock *sk;
+
+               res = inet_ctl_sock_create(&sk, PF_INET, SOCK_RAW,
+                                          IPPROTO_TCP, net);
+               if (res)
+                       goto fail;
+               *per_cpu_ptr(net->ipv4.tcp_sk, cpu) = sk;
+       }
         net->ipv4.sysctl_tcp_ecn = 2;
         return 0;
-}
  
-static void __net_exit tcp_sk_exit(struct net *net)
-{
+fail:
+       tcp_sk_exit(net);
+
+       return res;
  }
  
  static void __net_exit tcp_sk_exit_batch(struct list_head *net_exit_list)
author	Eric Dumazet <edumazet@google.com>
	Fri, 30 Jan 2015 05:35:05 +0000 (21:35 -0800)
committer	Luis Henriques <luis.henriques@canonical.com>
	Tue, 10 Feb 2015 13:38:50 +0000 (13:38 +0000)
include/net/ip.h		patch \| blob \| blame \| history
include/net/netns/ipv4.h		patch \| blob \| blame \| history
net/ipv4/ip_output.c		patch \| blob \| blame \| history
net/ipv4/tcp_ipv4.c		patch \| blob \| blame \| history