2008-02-08 00:49:26 -05:00
|
|
|
/*
|
|
|
|
* IPv6 Syncookies implementation for the Linux kernel
|
|
|
|
*
|
|
|
|
* Authors:
|
|
|
|
* Glenn Griffin <ggriffin.kernel@gmail.com>
|
|
|
|
*
|
|
|
|
* Based on IPv4 implementation by Andi Kleen
|
|
|
|
* linux/net/ipv4/syncookies.c
|
|
|
|
*
|
|
|
|
* This program is free software; you can redistribute it and/or
|
|
|
|
* modify it under the terms of the GNU General Public License
|
|
|
|
* as published by the Free Software Foundation; either version
|
|
|
|
* 2 of the License, or (at your option) any later version.
|
|
|
|
*
|
|
|
|
*/
|
|
|
|
|
|
|
|
#include <linux/tcp.h>
|
|
|
|
#include <linux/random.h>
|
|
|
|
#include <linux/cryptohash.h>
|
|
|
|
#include <linux/kernel.h>
|
|
|
|
#include <net/ipv6.h>
|
|
|
|
#include <net/tcp.h>
|
|
|
|
|
|
|
|
extern int sysctl_tcp_syncookies;
|
2008-03-24 01:21:28 -04:00
|
|
|
extern __u32 syncookie_secret[2][16-4+SHA_DIGEST_WORDS];
|
2008-02-08 00:49:26 -05:00
|
|
|
|
|
|
|
#define COOKIEBITS 24 /* Upper bits store count */
|
|
|
|
#define COOKIEMASK (((__u32)1 << COOKIEBITS) - 1)
|
|
|
|
|
|
|
|
/*
|
|
|
|
* This table has to be sorted and terminated with (__u16)-1.
|
|
|
|
* XXX generate a better table.
|
|
|
|
* Unresolved Issues: HIPPI with a 64k MSS is not well supported.
|
|
|
|
*
|
|
|
|
* Taken directly from ipv4 implementation.
|
|
|
|
* Should this list be modified for ipv6 use or is it close enough?
|
|
|
|
* rfc 2460 8.3 suggests mss values 20 bytes less than ipv4 counterpart
|
|
|
|
*/
|
|
|
|
static __u16 const msstab[] = {
|
|
|
|
64 - 1,
|
|
|
|
256 - 1,
|
|
|
|
512 - 1,
|
|
|
|
536 - 1,
|
|
|
|
1024 - 1,
|
|
|
|
1440 - 1,
|
|
|
|
1460 - 1,
|
|
|
|
4312 - 1,
|
|
|
|
(__u16)-1
|
|
|
|
};
|
|
|
|
/* The number doesn't include the -1 terminator */
|
|
|
|
#define NUM_MSS (ARRAY_SIZE(msstab) - 1)
|
|
|
|
|
|
|
|
/*
|
|
|
|
* This (misnamed) value is the age of syncookie which is permitted.
|
|
|
|
* Its ideal value should be dependent on TCP_TIMEOUT_INIT and
|
|
|
|
* sysctl_tcp_retries1. It's a rather complicated formula (exponential
|
|
|
|
* backoff) to compute at runtime so it's currently hardcoded here.
|
|
|
|
*/
|
|
|
|
#define COUNTER_TRIES 4
|
|
|
|
|
|
|
|
static inline struct sock *get_cookie_sock(struct sock *sk, struct sk_buff *skb,
|
|
|
|
struct request_sock *req,
|
|
|
|
struct dst_entry *dst)
|
|
|
|
{
|
|
|
|
struct inet_connection_sock *icsk = inet_csk(sk);
|
|
|
|
struct sock *child;
|
|
|
|
|
|
|
|
child = icsk->icsk_af_ops->syn_recv_sock(sk, skb, req, dst);
|
|
|
|
if (child)
|
|
|
|
inet_csk_reqsk_queue_add(sk, req, child);
|
|
|
|
else
|
|
|
|
reqsk_free(req);
|
|
|
|
|
|
|
|
return child;
|
|
|
|
}
|
|
|
|
|
2009-06-24 02:13:48 -04:00
|
|
|
static DEFINE_PER_CPU(__u32 [16 + 5 + SHA_WORKSPACE_WORDS],
|
|
|
|
ipv6_cookie_scratch);
|
2008-02-08 00:49:26 -05:00
|
|
|
|
|
|
|
static u32 cookie_hash(struct in6_addr *saddr, struct in6_addr *daddr,
|
|
|
|
__be16 sport, __be16 dport, u32 count, int c)
|
|
|
|
{
|
2009-06-24 02:13:48 -04:00
|
|
|
__u32 *tmp = __get_cpu_var(ipv6_cookie_scratch);
|
2008-02-08 00:49:26 -05:00
|
|
|
|
|
|
|
/*
|
|
|
|
* we have 320 bits of information to hash, copy in the remaining
|
|
|
|
* 192 bits required for sha_transform, from the syncookie_secret
|
|
|
|
* and overwrite the digest with the secret
|
|
|
|
*/
|
|
|
|
memcpy(tmp + 10, syncookie_secret[c], 44);
|
|
|
|
memcpy(tmp, saddr, 16);
|
|
|
|
memcpy(tmp + 4, daddr, 16);
|
|
|
|
tmp[8] = ((__force u32)sport << 16) + (__force u32)dport;
|
|
|
|
tmp[9] = count;
|
|
|
|
sha_transform(tmp + 16, (__u8 *)tmp, tmp + 16 + 5);
|
|
|
|
|
|
|
|
return tmp[17];
|
|
|
|
}
|
|
|
|
|
|
|
|
static __u32 secure_tcp_syn_cookie(struct in6_addr *saddr, struct in6_addr *daddr,
|
|
|
|
__be16 sport, __be16 dport, __u32 sseq,
|
|
|
|
__u32 count, __u32 data)
|
|
|
|
{
|
|
|
|
return (cookie_hash(saddr, daddr, sport, dport, 0, 0) +
|
|
|
|
sseq + (count << COOKIEBITS) +
|
|
|
|
((cookie_hash(saddr, daddr, sport, dport, count, 1) + data)
|
|
|
|
& COOKIEMASK));
|
|
|
|
}
|
|
|
|
|
|
|
|
static __u32 check_tcp_syn_cookie(__u32 cookie, struct in6_addr *saddr,
|
|
|
|
struct in6_addr *daddr, __be16 sport,
|
|
|
|
__be16 dport, __u32 sseq, __u32 count,
|
|
|
|
__u32 maxdiff)
|
|
|
|
{
|
|
|
|
__u32 diff;
|
|
|
|
|
|
|
|
cookie -= cookie_hash(saddr, daddr, sport, dport, 0, 0) + sseq;
|
|
|
|
|
|
|
|
diff = (count - (cookie >> COOKIEBITS)) & ((__u32) -1 >> COOKIEBITS);
|
|
|
|
if (diff >= maxdiff)
|
|
|
|
return (__u32)-1;
|
|
|
|
|
|
|
|
return (cookie -
|
|
|
|
cookie_hash(saddr, daddr, sport, dport, count - diff, 1))
|
|
|
|
& COOKIEMASK;
|
|
|
|
}
|
|
|
|
|
|
|
|
__u32 cookie_v6_init_sequence(struct sock *sk, struct sk_buff *skb, __u16 *mssp)
|
|
|
|
{
|
|
|
|
struct ipv6hdr *iph = ipv6_hdr(skb);
|
|
|
|
const struct tcphdr *th = tcp_hdr(skb);
|
|
|
|
int mssind;
|
|
|
|
const __u16 mss = *mssp;
|
|
|
|
|
2009-04-19 05:43:48 -04:00
|
|
|
tcp_synq_overflow(sk);
|
2008-02-08 00:49:26 -05:00
|
|
|
|
|
|
|
for (mssind = 0; mss > msstab[mssind + 1]; mssind++)
|
|
|
|
;
|
|
|
|
*mssp = msstab[mssind] + 1;
|
|
|
|
|
2008-07-16 23:31:16 -04:00
|
|
|
NET_INC_STATS_BH(sock_net(sk), LINUX_MIB_SYNCOOKIESSENT);
|
2008-02-08 00:49:26 -05:00
|
|
|
|
|
|
|
return secure_tcp_syn_cookie(&iph->saddr, &iph->daddr, th->source,
|
|
|
|
th->dest, ntohl(th->seq),
|
|
|
|
jiffies / (HZ * 60), mssind);
|
|
|
|
}
|
|
|
|
|
|
|
|
static inline int cookie_check(struct sk_buff *skb, __u32 cookie)
|
|
|
|
{
|
|
|
|
struct ipv6hdr *iph = ipv6_hdr(skb);
|
|
|
|
const struct tcphdr *th = tcp_hdr(skb);
|
|
|
|
__u32 seq = ntohl(th->seq) - 1;
|
|
|
|
__u32 mssind = check_tcp_syn_cookie(cookie, &iph->saddr, &iph->daddr,
|
|
|
|
th->source, th->dest, seq,
|
|
|
|
jiffies / (HZ * 60), COUNTER_TRIES);
|
|
|
|
|
|
|
|
return mssind < NUM_MSS ? msstab[mssind] + 1 : 0;
|
|
|
|
}
|
|
|
|
|
|
|
|
struct sock *cookie_v6_check(struct sock *sk, struct sk_buff *skb)
|
|
|
|
{
|
2009-12-02 13:25:27 -05:00
|
|
|
struct tcp_options_received tcp_opt;
|
|
|
|
u8 *hash_location;
|
2008-02-08 00:49:26 -05:00
|
|
|
struct inet_request_sock *ireq;
|
|
|
|
struct inet6_request_sock *ireq6;
|
|
|
|
struct tcp_request_sock *treq;
|
|
|
|
struct ipv6_pinfo *np = inet6_sk(sk);
|
|
|
|
struct tcp_sock *tp = tcp_sk(sk);
|
|
|
|
const struct tcphdr *th = tcp_hdr(skb);
|
|
|
|
__u32 cookie = ntohl(th->ack_seq) - 1;
|
|
|
|
struct sock *ret = sk;
|
|
|
|
struct request_sock *req;
|
|
|
|
int mss;
|
|
|
|
struct dst_entry *dst;
|
|
|
|
__u8 rcv_wscale;
|
|
|
|
|
|
|
|
if (!sysctl_tcp_syncookies || !th->ack)
|
|
|
|
goto out;
|
|
|
|
|
2009-04-19 05:43:48 -04:00
|
|
|
if (tcp_synq_no_recent_overflow(sk) ||
|
2008-02-08 00:49:26 -05:00
|
|
|
(mss = cookie_check(skb, cookie)) == 0) {
|
2008-07-16 23:31:16 -04:00
|
|
|
NET_INC_STATS_BH(sock_net(sk), LINUX_MIB_SYNCOOKIESFAILED);
|
2008-02-08 00:49:26 -05:00
|
|
|
goto out;
|
|
|
|
}
|
|
|
|
|
2008-07-16 23:31:16 -04:00
|
|
|
NET_INC_STATS_BH(sock_net(sk), LINUX_MIB_SYNCOOKIESRECV);
|
2008-02-08 00:49:26 -05:00
|
|
|
|
tcp: Revert per-route SACK/DSACK/TIMESTAMP changes.
It creates a regression, triggering badness for SYN_RECV
sockets, for example:
[19148.022102] Badness at net/ipv4/inet_connection_sock.c:293
[19148.022570] NIP: c02a0914 LR: c02a0904 CTR: 00000000
[19148.023035] REGS: eeecbd30 TRAP: 0700 Not tainted (2.6.32)
[19148.023496] MSR: 00029032 <EE,ME,CE,IR,DR> CR: 24002442 XER: 00000000
[19148.024012] TASK = eee9a820[1756] 'privoxy' THREAD: eeeca000
This is likely caused by the change in the 'estab' parameter
passed to tcp_parse_options() when invoked by the functions
in net/ipv4/tcp_minisocks.c
But even if that is fixed, the ->conn_request() changes made in
this patch series is fundamentally wrong. They try to use the
listening socket's 'dst' to probe the route settings. The
listening socket doesn't even have a route, and you can't
get the right route (the child request one) until much later
after we setup all of the state, and it must be done by hand.
This stuff really isn't ready, so the best thing to do is a
full revert. This reverts the following commits:
f55017a93f1a74d50244b1254b9a2bd7ac9bbf7d
022c3f7d82f0f1c68018696f2f027b87b9bb45c2
1aba721eba1d84a2defce45b950272cee1e6c72a
cda42ebd67ee5fdf09d7057b5a4584d36fe8a335
345cda2fd695534be5a4494f1b59da9daed33663
dc343475ed062e13fc260acccaab91d7d80fd5b2
05eaade2782fb0c90d3034fd7a7d5a16266182bb
6a2a2d6bf8581216e08be15fcb563cfd6c430e1e
Signed-off-by: David S. Miller <davem@davemloft.net>
2009-12-15 23:56:42 -05:00
|
|
|
/* check for timestamp cookie support */
|
|
|
|
memset(&tcp_opt, 0, sizeof(tcp_opt));
|
|
|
|
tcp_parse_options(skb, &tcp_opt, &hash_location, 0);
|
|
|
|
|
|
|
|
if (tcp_opt.saw_tstamp)
|
|
|
|
cookie_check_timestamp(&tcp_opt);
|
|
|
|
|
2008-02-08 00:49:26 -05:00
|
|
|
ret = NULL;
|
|
|
|
req = inet6_reqsk_alloc(&tcp6_request_sock_ops);
|
|
|
|
if (!req)
|
|
|
|
goto out;
|
|
|
|
|
|
|
|
ireq = inet_rsk(req);
|
|
|
|
ireq6 = inet6_rsk(req);
|
|
|
|
treq = tcp_rsk(req);
|
|
|
|
|
2008-08-03 21:13:44 -04:00
|
|
|
if (security_inet_conn_request(sk, skb, req))
|
|
|
|
goto out_free;
|
2008-02-08 00:49:26 -05:00
|
|
|
|
|
|
|
req->mss = mss;
|
|
|
|
ireq->rmt_port = th->source;
|
2008-10-20 02:35:58 -04:00
|
|
|
ireq->loc_port = th->dest;
|
2008-02-08 00:49:26 -05:00
|
|
|
ipv6_addr_copy(&ireq6->rmt_addr, &ipv6_hdr(skb)->saddr);
|
|
|
|
ipv6_addr_copy(&ireq6->loc_addr, &ipv6_hdr(skb)->daddr);
|
|
|
|
if (ipv6_opt_accepted(sk, skb) ||
|
|
|
|
np->rxopt.bits.rxinfo || np->rxopt.bits.rxoinfo ||
|
|
|
|
np->rxopt.bits.rxhlim || np->rxopt.bits.rxohlim) {
|
|
|
|
atomic_inc(&skb->users);
|
|
|
|
ireq6->pktopts = skb;
|
|
|
|
}
|
|
|
|
|
|
|
|
ireq6->iif = sk->sk_bound_dev_if;
|
|
|
|
/* So that link locals have meaning */
|
|
|
|
if (!sk->sk_bound_dev_if &&
|
|
|
|
ipv6_addr_type(&ireq6->rmt_addr) & IPV6_ADDR_LINKLOCAL)
|
|
|
|
ireq6->iif = inet6_iif(skb);
|
|
|
|
|
|
|
|
req->expires = 0UL;
|
|
|
|
req->retrans = 0;
|
2008-07-26 05:21:54 -04:00
|
|
|
ireq->ecn_ok = 0;
|
tcp: Revert per-route SACK/DSACK/TIMESTAMP changes.
It creates a regression, triggering badness for SYN_RECV
sockets, for example:
[19148.022102] Badness at net/ipv4/inet_connection_sock.c:293
[19148.022570] NIP: c02a0914 LR: c02a0904 CTR: 00000000
[19148.023035] REGS: eeecbd30 TRAP: 0700 Not tainted (2.6.32)
[19148.023496] MSR: 00029032 <EE,ME,CE,IR,DR> CR: 24002442 XER: 00000000
[19148.024012] TASK = eee9a820[1756] 'privoxy' THREAD: eeeca000
This is likely caused by the change in the 'estab' parameter
passed to tcp_parse_options() when invoked by the functions
in net/ipv4/tcp_minisocks.c
But even if that is fixed, the ->conn_request() changes made in
this patch series is fundamentally wrong. They try to use the
listening socket's 'dst' to probe the route settings. The
listening socket doesn't even have a route, and you can't
get the right route (the child request one) until much later
after we setup all of the state, and it must be done by hand.
This stuff really isn't ready, so the best thing to do is a
full revert. This reverts the following commits:
f55017a93f1a74d50244b1254b9a2bd7ac9bbf7d
022c3f7d82f0f1c68018696f2f027b87b9bb45c2
1aba721eba1d84a2defce45b950272cee1e6c72a
cda42ebd67ee5fdf09d7057b5a4584d36fe8a335
345cda2fd695534be5a4494f1b59da9daed33663
dc343475ed062e13fc260acccaab91d7d80fd5b2
05eaade2782fb0c90d3034fd7a7d5a16266182bb
6a2a2d6bf8581216e08be15fcb563cfd6c430e1e
Signed-off-by: David S. Miller <davem@davemloft.net>
2009-12-15 23:56:42 -05:00
|
|
|
ireq->snd_wscale = tcp_opt.snd_wscale;
|
|
|
|
ireq->rcv_wscale = tcp_opt.rcv_wscale;
|
|
|
|
ireq->sack_ok = tcp_opt.sack_ok;
|
|
|
|
ireq->wscale_ok = tcp_opt.wscale_ok;
|
|
|
|
ireq->tstamp_ok = tcp_opt.saw_tstamp;
|
|
|
|
req->ts_recent = tcp_opt.saw_tstamp ? tcp_opt.rcv_tsval : 0;
|
2008-02-08 00:49:26 -05:00
|
|
|
treq->rcv_isn = ntohl(th->seq) - 1;
|
|
|
|
treq->snt_isn = cookie;
|
|
|
|
|
|
|
|
/*
|
|
|
|
* We need to lookup the dst_entry to get the correct window size.
|
|
|
|
* This is taken from tcp_v6_syn_recv_sock. Somebody please enlighten
|
|
|
|
* me if there is a preferred way.
|
|
|
|
*/
|
|
|
|
{
|
|
|
|
struct in6_addr *final_p = NULL, final;
|
|
|
|
struct flowi fl;
|
|
|
|
memset(&fl, 0, sizeof(fl));
|
|
|
|
fl.proto = IPPROTO_TCP;
|
|
|
|
ipv6_addr_copy(&fl.fl6_dst, &ireq6->rmt_addr);
|
|
|
|
if (np->opt && np->opt->srcrt) {
|
|
|
|
struct rt0_hdr *rt0 = (struct rt0_hdr *) np->opt->srcrt;
|
|
|
|
ipv6_addr_copy(&final, &fl.fl6_dst);
|
|
|
|
ipv6_addr_copy(&fl.fl6_dst, rt0->addr);
|
|
|
|
final_p = &final;
|
|
|
|
}
|
|
|
|
ipv6_addr_copy(&fl.fl6_src, &ireq6->loc_addr);
|
|
|
|
fl.oif = sk->sk_bound_dev_if;
|
2009-10-05 04:24:16 -04:00
|
|
|
fl.mark = sk->sk_mark;
|
2008-02-08 00:49:26 -05:00
|
|
|
fl.fl_ip_dport = inet_rsk(req)->rmt_port;
|
2009-10-15 02:30:45 -04:00
|
|
|
fl.fl_ip_sport = inet_sk(sk)->inet_sport;
|
2008-02-08 00:49:26 -05:00
|
|
|
security_req_classify_flow(req, &fl);
|
2008-08-03 21:13:44 -04:00
|
|
|
if (ip6_dst_lookup(sk, &dst, &fl))
|
|
|
|
goto out_free;
|
|
|
|
|
2008-02-08 00:49:26 -05:00
|
|
|
if (final_p)
|
|
|
|
ipv6_addr_copy(&fl.fl6_dst, final_p);
|
2008-11-25 20:35:18 -05:00
|
|
|
if ((xfrm_lookup(sock_net(sk), &dst, &fl, sk, 0)) < 0)
|
2008-08-03 21:13:44 -04:00
|
|
|
goto out_free;
|
2008-02-08 00:49:26 -05:00
|
|
|
}
|
|
|
|
|
2008-04-10 06:12:40 -04:00
|
|
|
req->window_clamp = tp->window_clamp ? :dst_metric(dst, RTAX_WINDOW);
|
2008-02-08 00:49:26 -05:00
|
|
|
tcp_select_initial_window(tcp_full_space(sk), req->mss,
|
|
|
|
&req->rcv_wnd, &req->window_clamp,
|
2008-04-10 06:12:40 -04:00
|
|
|
ireq->wscale_ok, &rcv_wscale);
|
2008-02-08 00:49:26 -05:00
|
|
|
|
|
|
|
ireq->rcv_wscale = rcv_wscale;
|
|
|
|
|
|
|
|
ret = get_cookie_sock(sk, skb, req, dst);
|
2008-08-03 21:13:44 -04:00
|
|
|
out:
|
|
|
|
return ret;
|
|
|
|
out_free:
|
|
|
|
reqsk_free(req);
|
|
|
|
return NULL;
|
2008-02-08 00:49:26 -05:00
|
|
|
}
|
|
|
|
|