diff -rpu linux-3.18.1/include/linux/tcp.h linux-3.18.1_AccECN/include/linux/tcp.h --- linux-3.18.1/include/linux/tcp.h 2014-12-16 18:39:45.000000000 +0100 +++ linux-3.18.1_AccECN/include/linux/tcp.h 2015-06-17 13:10:31.412039565 +0200 @@ -204,6 +204,14 @@ struct tcp_sock { u16 urg_data; /* Saved octet of OOB data and control flags */ u8 ecn_flags; /* ECN status bits. */ + u32 ce_cnt; /* C1 counter for CE marks at sender */ + u32 ect0_cnt; /* counter for ECT(0) marks at sender */ + u32 ect1_cnt; /* E1 counter for ECT(1) marks at sender */ + u32 nect_cnt; /* counter for non-ECT packets at sender */ + u32 ce_rec; /* C1 counter for CE marks at receiver */ + u32 ect0_rec; /* counter for ECT(0) marks at receiver */ + u32 ect1_rec; /* E1 counter for ECT(1) marks at receiver */ + u32 nect_rec; /* counter for non-ECT packets at receiver */ u8 reordering; /* Packet reordering metric. */ u32 snd_up; /* Urgent pointer */ diff -rpu linux-3.18.1/include/net/inet_sock.h linux-3.18.1_AccECN/include/net/inet_sock.h --- linux-3.18.1/include/net/inet_sock.h 2014-12-16 18:39:45.000000000 +0100 +++ linux-3.18.1_AccECN/include/net/inet_sock.h 2015-06-02 18:00:29.677717590 +0200 @@ -85,6 +85,7 @@ struct inet_request_sock { sack_ok : 1, wscale_ok : 1, ecn_ok : 1, + accecn_ok : 1, acked : 1, no_srccheck: 1; kmemcheck_bitfield_end(flags); diff -rpu linux-3.18.1/include/net/tcp.h linux-3.18.1_AccECN/include/net/tcp.h --- linux-3.18.1/include/net/tcp.h 2014-12-16 18:39:45.000000000 +0100 +++ linux-3.18.1_AccECN/include/net/tcp.h 2015-06-17 13:16:01.728027383 +0200 @@ -387,10 +387,15 @@ static inline void tcp_dec_quickack_mode } } -#define TCP_ECN_OK 1 -#define TCP_ECN_QUEUE_CWR 2 -#define TCP_ECN_DEMAND_CWR 4 -#define TCP_ECN_SEEN 8 +#define TCP_ECN_OK 0x01 +#define TCP_ECN_QUEUE_CWR 0x02 +#define TCP_ECN_DEMAND_CWR 0x04 +#define TCP_ECN_SEEN 0x08 + +#define TCP_ACCECN_OK 0x10 +#define TCP_ACCECN_CE 0x20 +#define TCP_ACCECN_ECT1 0x40 +#define TCP_ACCECN_NECT 0x80 enum tcp_tw_status { TCP_TW_SUCCESS = 0, @@ -685,6 +690,8 @@ static inline u32 tcp_skb_timestamp(cons #define TCPHDR_ECE 0x40 #define TCPHDR_CWR 0x80 +#define TCPHDR_NS 0x01 + /* This is what the send packet queuing engine uses to pass * TCP per-packet control information to the transmission code. * We also store the host-order sequence numbers in here too. @@ -705,6 +712,7 @@ struct tcp_skb_cb { __u32 tcp_gso_segs; }; __u8 tcp_flags; /* TCP header flags. (tcp[13]) */ + __u8 tcp_flags2; /* TCP header flags. (tcp[12]) */ __u8 sacked; /* State flags for SACK/FACK. */ #define TCPCB_SACKED_ACKED 0x01 /* SKB ACK'd by a SACK block */ @@ -1135,6 +1143,7 @@ static inline void tcp_openreq_init(stru ireq->wscale_ok = rx_opt->wscale_ok; ireq->acked = 0; ireq->ecn_ok = 0; + ireq->accecn_ok = 0; ireq->ir_rmt_port = tcp_hdr(skb)->source; ireq->ir_num = ntohs(tcp_hdr(skb)->dest); ireq->ir_mark = inet_request_mark(sk, skb); diff -rpu linux-3.18.1/include/uapi/linux/tcp.h linux-3.18.1_AccECN/include/uapi/linux/tcp.h --- linux-3.18.1/include/uapi/linux/tcp.h 2014-12-16 18:39:45.000000000 +0100 +++ linux-3.18.1_AccECN/include/uapi/linux/tcp.h 2015-06-02 16:49:55.905873732 +0200 @@ -27,7 +27,8 @@ struct tcphdr { __be32 seq; __be32 ack_seq; #if defined(__LITTLE_ENDIAN_BITFIELD) - __u16 res1:4, + __u16 ns:1, + res1:3, doff:4, fin:1, syn:1, @@ -39,7 +40,8 @@ struct tcphdr { cwr:1; #elif defined(__BIG_ENDIAN_BITFIELD) __u16 doff:4, - res1:4, + res1:3, + ns:1, cwr:1, ece:1, urg:1, diff -rpu linux-3.18.1/net/ipv4/syncookies.c linux-3.18.1_AccECN/net/ipv4/syncookies.c --- linux-3.18.1/net/ipv4/syncookies.c 2014-12-16 18:39:45.000000000 +0100 +++ linux-3.18.1_AccECN/net/ipv4/syncookies.c 2015-06-02 18:00:59.845716477 +0200 @@ -270,6 +270,7 @@ struct sock *cookie_v4_check(struct sock struct rtable *rt; __u8 rcv_wscale; bool ecn_ok = false; + bool accecn_ok = false; struct flowi4 fl4; if (!sysctl_tcp_syncookies || !th->ack || th->rst) @@ -306,6 +307,7 @@ struct sock *cookie_v4_check(struct sock ireq->ir_rmt_addr = ip_hdr(skb)->saddr; ireq->ir_mark = inet_request_mark(sk, skb); ireq->ecn_ok = ecn_ok; + ireq->accecn_ok = accecn_ok; ireq->snd_wscale = tcp_opt.snd_wscale; ireq->sack_ok = tcp_opt.sack_ok; ireq->wscale_ok = tcp_opt.wscale_ok; diff -rpu linux-3.18.1/net/ipv4/tcp_input.c linux-3.18.1_AccECN/net/ipv4/tcp_input.c --- linux-3.18.1/net/ipv4/tcp_input.c 2014-12-16 18:39:45.000000000 +0100 +++ linux-3.18.1_AccECN/net/ipv4/tcp_input.c 2015-06-17 13:08:22.672044313 +0200 @@ -222,6 +222,8 @@ static void __tcp_ecn_check_ce(struct tc { switch (TCP_SKB_CB(skb)->ip_dsfield & INET_ECN_MASK) { case INET_ECN_NOT_ECT: + tp->nect_rec++; + tp->ecn_flags |= TCP_ACCECN_NECT; /* Funny extension: if ECT is not set on a segment, * and we already seen ECT on a previous segment, * it is probably a retransmit. @@ -230,6 +232,8 @@ static void __tcp_ecn_check_ce(struct tc tcp_enter_quickack_mode((struct sock *)tp); break; case INET_ECN_CE: + tp->ce_rec++; + tp->ecn_flags |= TCP_ACCECN_CE; if (tcp_ca_needs_ecn((struct sock *)tp)) tcp_ca_event((struct sock *)tp, CA_EVENT_ECN_IS_CE); @@ -240,7 +244,15 @@ static void __tcp_ecn_check_ce(struct tc } tp->ecn_flags |= TCP_ECN_SEEN; break; - default: + case INET_ECN_ECT_0: + tp->ect0_rec++; + if (tcp_ca_needs_ecn((struct sock *)tp)) + tcp_ca_event((struct sock *)tp, CA_EVENT_ECN_NO_CE); + tp->ecn_flags |= TCP_ECN_SEEN; + break; + case INET_ECN_ECT_1: + tp->ect1_rec++; + tp->ecn_flags |= TCP_ACCECN_ECT1; if (tcp_ca_needs_ecn((struct sock *)tp)) tcp_ca_event((struct sock *)tp, CA_EVENT_ECN_NO_CE); tp->ecn_flags |= TCP_ECN_SEEN; @@ -256,20 +268,59 @@ static void tcp_ecn_check_ce(struct tcp_ static void tcp_ecn_rcv_synack(struct tcp_sock *tp, const struct tcphdr *th) { - if ((tp->ecn_flags & TCP_ECN_OK) && (!th->ece || th->cwr)) + if ((tp->ecn_flags & TCP_ECN_OK) && !(tp->ecn_flags & TCP_ACCECN_OK) + && (!th->ece || th->cwr)) tp->ecn_flags &= ~TCP_ECN_OK; + if ((tp->ecn_flags & TCP_ACCECN_OK) && (th->ece || !th->cwr || th->ns)) { + tp->ecn_flags &= ~TCP_ECN_OK; + tp->ecn_flags &= ~TCP_ACCECN_OK; + } } static void tcp_ecn_rcv_syn(struct tcp_sock *tp, const struct tcphdr *th) { if ((tp->ecn_flags & TCP_ECN_OK) && (!th->ece || !th->cwr)) tp->ecn_flags &= ~TCP_ECN_OK; + if ((tp->ecn_flags & TCP_ACCECN_OK) && (!th->ns)) + tp->ecn_flags &= ~TCP_ACCECN_OK; } -static bool tcp_ecn_rcv_ecn_echo(const struct tcp_sock *tp, const struct tcphdr *th) +static bool tcp_ecn_rcv_ecn_echo(struct tcp_sock *tp, const struct tcphdr *th) { - if (th->ece && !th->syn && (tp->ecn_flags & TCP_ECN_OK)) - return true; + + if (tp->ecn_flags & TCP_ECN_OK && !th->syn) { + if (tp->ecn_flags & TCP_ACCECN_OK) { + /* check ACE codepoint and increase counters */ + u8 tmp_cnt = 0; + if (th->ns) { + if (th->ece && th->cwr) { + tp->nect_cnt++; + return false; + } else if (th->ece) + tmp_cnt = 1; + else if (th->cwr) + tmp_cnt = 2; + tp->ect1_cnt += ((tmp_cnt+3 - (tp->ect1_cnt % 3)))%3; + return false; + } else { + if (th->ece && th->cwr) + tmp_cnt = 3; + else if (th->ece) + tmp_cnt = 1; + else if (th->cwr) + tmp_cnt = 2; + u8 tmp_add = ((tmp_cnt+4 - (tp->ce_cnt % 4)))%4; + printk("C1-sender: %u tmp_cnt: %u\n", tp->ce_cnt, tmp_cnt); + tp->ce_cnt += tmp_add; + if (tmp_add) + /* indicate congestion if C1 counter is increased */ + return true; + return false; + } + } else if (th->ece) { + return true; + } + } return false; } @@ -3457,8 +3508,14 @@ static int tcp_ack(struct sock *sk, cons tcp_update_wl(tp, ack_seq); tp->snd_una = ack; flag |= FLAG_WIN_UPDATE; + u32 ack_ev_flags = CA_ACK_WIN_UPDATE; + + if (tcp_ecn_rcv_ecn_echo(tp, tcp_hdr(skb))) { + flag |= FLAG_ECE; + ack_ev_flags |= CA_ACK_ECE; + } - tcp_in_ack_event(sk, CA_ACK_WIN_UPDATE); + tcp_in_ack_event(sk, ack_ev_flags); NET_INC_STATS_BH(sock_net(sk), LINUX_MIB_TCPHPACKS); } else { @@ -5883,6 +5940,7 @@ static void tcp_ecn_create_request(struc const struct net *net = sock_net(listen_sk); bool th_ecn = th->ece && th->cwr; bool ect, need_ecn; + bool th_ns = th->ns; if (!th_ecn) return; @@ -5890,9 +5948,11 @@ static void tcp_ecn_create_request(struc ect = !INET_ECN_is_not_ect(TCP_SKB_CB(skb)->ip_dsfield); need_ecn = tcp_ca_needs_ecn(listen_sk); - if (!ect && !need_ecn && net->ipv4.sysctl_tcp_ecn) + if (!ect && !need_ecn && net->ipv4.sysctl_tcp_ecn) { inet_rsk(req)->ecn_ok = 1; - else if (ect && need_ecn) + if (th_ns && net->ipv4.sysctl_tcp_ecn==4) + inet_rsk(req)->accecn_ok = 1; + } else if (ect && need_ecn) inet_rsk(req)->ecn_ok = 1; } diff -rpu linux-3.18.1/net/ipv4/tcp_minisocks.c linux-3.18.1_AccECN/net/ipv4/tcp_minisocks.c --- linux-3.18.1/net/ipv4/tcp_minisocks.c 2014-12-16 18:39:45.000000000 +0100 +++ linux-3.18.1_AccECN/net/ipv4/tcp_minisocks.c 2015-06-10 19:05:39.296256303 +0200 @@ -397,6 +397,7 @@ static void tcp_ecn_openreq_child(struct const struct request_sock *req) { tp->ecn_flags = inet_rsk(req)->ecn_ok ? TCP_ECN_OK : 0; + tp->ecn_flags |= inet_rsk(req)->accecn_ok ? TCP_ACCECN_OK : 0; } /* This is not only more efficient than what we used to do, it eliminates diff -rpu linux-3.18.1/net/ipv4/tcp_output.c linux-3.18.1_AccECN/net/ipv4/tcp_output.c --- linux-3.18.1/net/ipv4/tcp_output.c 2014-12-16 18:39:45.000000000 +0100 +++ linux-3.18.1_AccECN/net/ipv4/tcp_output.c 2015-06-17 13:05:47.468050037 +0200 @@ -327,6 +327,13 @@ static void tcp_ecn_send_synack(struct s TCP_SKB_CB(skb)->tcp_flags &= ~TCPHDR_ECE; else if (tcp_ca_needs_ecn(sk)) INET_ECN_xmit(sk); + + if (tp->ecn_flags & TCP_ACCECN_OK) { + /* negotiate AccECN instead */ + TCP_SKB_CB(skb)->tcp_flags &= ~TCPHDR_ECE; + TCP_SKB_CB(skb)->tcp_flags |= TCPHDR_CWR; + TCP_SKB_CB(skb)->tcp_flags2 &= ~TCPHDR_NS; + } } /* Packet ECN state for a SYN. */ @@ -334,11 +341,21 @@ static void tcp_ecn_send_syn(struct sock { struct tcp_sock *tp = tcp_sk(sk); + tp->ce_cnt = 0; + tp->ect0_cnt = 0; + tp->ect1_cnt = 0; + tp->nect_cnt = 0; tp->ecn_flags = 0; if (sock_net(sk)->ipv4.sysctl_tcp_ecn == 1 || + sock_net(sk)->ipv4.sysctl_tcp_ecn == 4 || tcp_ca_needs_ecn(sk)) { TCP_SKB_CB(skb)->tcp_flags |= TCPHDR_ECE | TCPHDR_CWR; tp->ecn_flags = TCP_ECN_OK; + if (sock_net(sk)->ipv4.sysctl_tcp_ecn == 4) { + /* request AccECN */ + TCP_SKB_CB(skb)->tcp_flags2 |= TCPHDR_NS; + tp->ecn_flags |= TCP_ACCECN_OK; + } if (tcp_ca_needs_ecn(sk)) INET_ECN_xmit(sk); } @@ -353,6 +370,62 @@ tcp_ecn_make_synack(const struct request if (tcp_ca_needs_ecn(sk)) INET_ECN_xmit(sk); } + if (inet_rsk(req)->accecn_ok) { + /* negotiate AccECN instead */ + th->ece = 0; + th->cwr = 1; + th->ns = 0; + } +} + +/* set the ACE codepoint with respect to the C1 counter */ +static void tcp_accecn_signal_c1(struct sock *sk, struct sk_buff *skb) +{ + struct tcp_sock *tp = tcp_sk(sk); + struct tcphdr *th = tcp_hdr(skb); + + th->ns = 0; + switch (tp->ce_rec % 4) { + case 0: + th->ece = 0; + th->cwr = 0; + break; + case 1: + th->ece = 1; + th->cwr = 0; + break; + case 2: + th->ece = 0; + th->cwr = 1; + break; + case 3: + th->ece = 1; + th->cwr = 1; + break; + } +} + +/* set the ACE codepoint with respect to the E1 counter */ +static void tcp_accecn_signal_e1(struct sock *sk, struct sk_buff *skb) +{ + struct tcp_sock *tp = tcp_sk(sk); + struct tcphdr *th = tcp_hdr(skb); + + th->ns = 1; + switch (tp->ect1_rec % 3) { + case 0: + th->ece = 0; + th->cwr = 0; + break; + case 1: + th->ece = 1; + th->cwr = 0; + break; + case 2: + th->ece = 0; + th->cwr = 1; + break; + } } /* Set up ECN state for a packet on a ESTABLISHED socket that is about to @@ -362,6 +435,7 @@ static void tcp_ecn_send(struct sock *sk int tcp_header_len) { struct tcp_sock *tp = tcp_sk(sk); + struct tcphdr *th = tcp_hdr(skb); if (tp->ecn_flags & TCP_ECN_OK) { /* Not-retransmitted data segment: set ECT and inject CWR. */ @@ -377,8 +451,30 @@ static void tcp_ecn_send(struct sock *sk /* ACK or retransmitted segment: clear ECT|CE */ INET_ECN_dontxmit(sk); } - if (tp->ecn_flags & TCP_ECN_DEMAND_CWR) - tcp_hdr(skb)->ece = 1; + + if (tp->ecn_flags & TCP_ACCECN_OK) { + /* Set AccECN codepoint */ + /* (note that there is a priority order which to set first) + /* TODO: protentialy verify that all counters are frequently signaled */ + if (tp->ecn_flags & TCP_ACCECN_CE) { + tcp_accecn_signal_c1(sk, skb); + tp->ecn_flags &= ~TCP_ACCECN_CE; + } else if (tp->ecn_flags & TCP_ACCECN_ECT1) { + tcp_accecn_signal_e1(sk, skb); + tp->ecn_flags &= ~TCP_ACCECN_ECT1; + } else if (tp->ecn_flags & TCP_ACCECN_NECT) { + th->ece = 1; + th->cwr = 1; + th->ns = 1; + tp->ecn_flags &= ~TCP_ACCECN_NECT; + } else { + /* send CE counter by default */ + /* if no of the counter have changed */ + /* TODO: implement algo to send one or the other alternatively */ + tcp_accecn_signal_c1(sk, skb); + } + } else if (tp->ecn_flags & TCP_ECN_DEMAND_CWR) + th->ece = 1; } } @@ -950,7 +1046,8 @@ static int tcp_transmit_skb(struct sock th->seq = htonl(tcb->seq); th->ack_seq = htonl(tp->rcv_nxt); *(((__be16 *)th) + 6) = htons(((tcp_header_size >> 2) << 12) | - tcb->tcp_flags); + ((tcb->tcp_flags2 & 0x0F) << 8) | + tcb->tcp_flags ); if (unlikely(tcb->tcp_flags & TCPHDR_SYN)) { /* RFC1323: The window in SYN & SYN/ACK segments