From: Chia-Yu Chang
AccECN option may fail in various way, handle these:
- Remove option from SYN/ACK rexmits to handle blackholes
- If no option arrives in SYN/ACK, assume Option is not usable
- If an option arrives later, re-enabled
- If option is zeroed, disable AccECN option processing
This patch use existing padding bits in tcp_request_sock and
holes in tcp_sock without increasing the size.
Signed-off-by: Ilpo Järvinen
Signed-off-by: Chia-Yu Chang
---
include/linux/tcp.h | 4 ++-
include/net/tcp.h| 7 +
net/ipv4/tcp.c | 1 +
net/ipv4/tcp_input.c | 68 +++-
net/ipv4/tcp_minisocks.c | 38 ++
net/ipv4/tcp_output.c| 5 ++-
6 files changed, 113 insertions(+), 10 deletions(-)
diff --git a/include/linux/tcp.h b/include/linux/tcp.h
index 0740efaaef28..b5066eef8782 100644
--- a/include/linux/tcp.h
+++ b/include/linux/tcp.h
@@ -173,6 +173,7 @@ struct tcp_request_sock {
u8 syn_ect_snt: 2,
syn_ect_rcv: 2,
accecn_fail_mode:4;
+ u8 saw_accecn_opt :2;
#ifdef CONFIG_TCP_AO
u8 ao_keyid;
u8 ao_rcv_next;
@@ -409,7 +410,8 @@ struct tcp_sock {
syn_fastopen_child:1; /* created TFO passive child socket */
u8 keepalive_probes; /* num of allowed keep alive probes */
- u8 accecn_fail_mode:4; /* AccECN failure handling */
+ u8 accecn_fail_mode:4, /* AccECN failure handling */
+ saw_accecn_opt:2; /* An AccECN option was seen */
u32 tcp_tx_delay; /* delay (in usec) added to TX packets */
/* RTT measurement */
diff --git a/include/net/tcp.h b/include/net/tcp.h
index 3419618a7891..5e4593e39de4 100644
--- a/include/net/tcp.h
+++ b/include/net/tcp.h
@@ -279,6 +279,12 @@ static inline void tcp_accecn_fail_mode_set(struct
tcp_sock *tp, u8 mode)
tp->accecn_fail_mode |= mode;
}
+/* tp->saw_accecn_opt states */
+#define TCP_ACCECN_OPT_NOT_SEEN0x0
+#define TCP_ACCECN_OPT_EMPTY_SEEN 0x1
+#define TCP_ACCECN_OPT_COUNTER_SEEN0x2
+#define TCP_ACCECN_OPT_FAIL_SEEN 0x3
+
/* Flags in tp->nonagle */
#define TCP_NAGLE_OFF 1 /* Nagle's algo is disabled */
#define TCP_NAGLE_CORK 2 /* Socket is corked */
@@ -480,6 +486,7 @@ static inline int tcp_accecn_extract_syn_ect(u8 ace)
bool tcp_accecn_validate_syn_feedback(struct sock *sk, u8 ace, u8 sent_ect);
void tcp_accecn_third_ack(struct sock *sk, const struct sk_buff *skb,
u8 syn_ect_snt);
+u8 tcp_accecn_option_init(const struct sk_buff *skb, u8 opt_offset);
void tcp_ecn_received_counters(struct sock *sk, const struct sk_buff *skb,
u32 payload_len);
diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c
index 20a2e30e15f3..e68b9706eeff 100644
--- a/net/ipv4/tcp.c
+++ b/net/ipv4/tcp.c
@@ -3399,6 +3399,7 @@ int tcp_disconnect(struct sock *sk, int flags)
tp->delivered_ce = 0;
tp->wait_third_ack = 0;
tp->accecn_fail_mode = 0;
+ tp->saw_accecn_opt = TCP_ACCECN_OPT_NOT_SEEN;
tcp_accecn_init_counters(tp);
tp->prev_ecnfield = 0;
tp->accecn_opt_tstamp = 0;
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 10c37a41b9a5..c93e4bffb23e 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -446,9 +446,8 @@ bool tcp_accecn_validate_syn_feedback(struct sock *sk, u8
ace, u8 sent_ect)
}
/* See Table 2 of the AccECN draft */
-
-static void tcp_ecn_rcv_synack(struct sock *sk, const struct tcphdr *th,
- u8 ip_dsfield)
+static void tcp_ecn_rcv_synack(struct sock *sk, const struct sk_buff *skb,
+ const struct tcphdr *th, u8 ip_dsfield)
{
struct tcp_sock *tp = tcp_sk(sk);
u8 ace = tcp_accecn_ace(th);
@@ -487,7 +486,19 @@ static void tcp_ecn_rcv_synack(struct sock *sk, const
struct tcphdr *th,
default:
tcp_ecn_mode_set(tp, TCP_ECN_MODE_ACCECN);
tp->syn_ect_rcv = ip_dsfield & INET_ECN_MASK;
- tp->accecn_opt_demand = 2;
+ if (tp->rx_opt.accecn &&
+ tp->saw_accecn_opt < TCP_ACCECN_OPT_COUNTER_SEEN) {
+ u8 saw_opt = tcp_accecn_option_init(skb,
+ tp->rx_opt.accecn);
+
+ tp->saw_accecn_opt = saw_opt;
+ if (tp->saw_accecn_opt == TCP_ACCECN_OPT_FAIL_SEEN) {
+ u8 fail_mode = TCP_ACCECN_OPT_FAIL_RECV;
+
+ tcp_accecn_fail_mode_set(tp, fail_mode);
+ }
+ tp->accecn_opt_demand = 2;
+ }
if