~kris/9p

9hist

e34105fedb6735d39d0afa1982128fcd53016384 — David du Colombier 27 years ago 753958f
Plan 9 from Bell Labs 1999-04-01
1 files changed, 124 insertions(+), 69 deletions(-)

M ip/tcp.c
M ip/tcp.c => ip/tcp.c +124 -69
@@ 36,18 36,19 @@ enum
	MSS_LENGTH	= 4,		/* Mean segment size */
	MSL2		= 10,
	MSPTICK		= 50,		/* Milliseconds per timer tick */
	DEF_MSS		= 1024,		/* Default mean segment */
	DEF_RTT		= 150,		/* Default round trip */
	DEF_MSS		= 1460,		/* Default mean segment */
	DEF_RTT		= 1000,		/* Default round trip */
	DEF_KAT		= 10000,	/* Default keep alive trip in ms */
	TCP_LISTEN	= 0,		/* Listen connection */
	TCP_CONNECT	= 1,		/* Outgoing connection */

	TCPREXMTTHRESH  = 3,            /* dupack threshhold for rxt */

	FORCE		= 1,
	CLONE		= 2,
	RETRAN		= 4,
	ACTIVE		= 8,
	SYNACK		= 16,
	ACKED		= 32,

	LOGAGAIN	= 3,
	LOGDGAIN	= 2,


@@ 122,6 123,7 @@ struct	Tcp
	ushort	wnd;
	ushort	urg;
	ushort	mss;
        ushort  len;  /* size of data */
};

typedef struct Reseq Reseq;


@@ 150,6 152,11 @@ struct Tcpctl
		ulong	urg;		/* Urgent data pointer */
		ulong	wl1;
		ulong	wl2;

                /* to implement tahoe and reno TCP */
                ulong dupacks;          /* number of duplicate acks rcvd */
	        int   recovery;         /* loss recovery flag */
	        ulong rxt;              /* right window marker for recovery */
	} snd;
	struct {
		ulong	nxt;		/* Receive pointer to next uchar slot */


@@ 203,6 210,7 @@ struct Tcpstats
	ulong	tcpInSegs;
	ulong	tcpOutSegs;
	ulong	tcpRetransSegs;
        ulong   tcpRetransTimeouts;
	ulong	InErrs;
	ulong	OutRsts;
};


@@ 241,6 249,7 @@ void	tcpsndsyn(Tcpctl*);
void	tcprcvwin(Conv*);
void	tcpacktimer(void*);
void	tcpkeepalive(void*);
void    tcprxmit(Conv*, int);

void
tcpsetstate(Conv *s, uchar newstate)


@@ 575,6 584,22 @@ localclose(Conv *s, char *reason)	 /*  called with tcb locked */
	tcpsetstate(s, Closed);
}

/* mtu (- TCP + IP hdr len) of 1st hop */
int
tcpmtu(Conv *s)
{
	Ipifc *ifc;
	int mtu;

	mtu = 0;
	ifc = findipifc(s->p->f, s->raddr, 0);
	if(ifc != nil)
		mtu = ifc->maxmtu - ifc->m->hsize - (TCP_PKT + TCP_HDRSIZE);
	if(mtu < 4)
		mtu = DEF_MSS;
	return mtu;
}

void
inittcpctl(Conv *s)
{


@@ 585,10 610,7 @@ inittcpctl(Conv *s)

	memset(tcb, 0, sizeof(Tcpctl));

	tcb->cwind = tcp_mss;
	tcb->mss = tcp_mss;
	tcb->ssthresh = 65535;
	tcb->srtt = 0;

	tcb->timer.start = tcp_irtt / MSPTICK;
	tcb->timer.func = tcptimeout;


@@ 597,7 619,6 @@ inittcpctl(Conv *s)
	tcb->acktimer.start = TCP_ACK / MSPTICK;
	tcb->acktimer.func = tcpacktimer;
	tcb->acktimer.arg = s;
	tcb->kacounter = 0;
	tcb->katimer.start = DEF_KAT / MSPTICK;
	tcb->katimer.func = tcpkeepalive;
	tcb->katimer.arg = s;


@@ 612,22 633,8 @@ inittcpctl(Conv *s)
	hnputs(h->tcpdport, s->rport);
	v6tov4(h->tcpsrc, s->laddr);
	v6tov4(h->tcpdst, s->raddr);
}

/* mtu (- TCP + IP hdr len) of 1st hop */
int
tcpmtu(Conv *s)
{
	Ipifc *ifc;
	int mtu;

	mtu = 0;
	ifc = findipifc(s->p->f, s->raddr, 0);
	if(ifc != nil)
		mtu = ifc->maxmtu - ifc->m->hsize - (TCP_PKT + TCP_HDRSIZE);
	if(mtu < 4)
		mtu = DEF_MSS;
	return mtu;
	tcb->mss = tcb->cwind = tcpmtu(s);
}

void


@@ 779,6 786,7 @@ ntohtcp(Tcp *tcph, Block **bpp)
	tcph->wnd = nhgets(h->tcpwin);
	tcph->urg = nhgets(h->tcpurg);
	tcph->mss = 0;
	tcph->len = nhgets(h->length) - (hdrlen + TCP_PKT);

	*bpp = pullupblock(*bpp, hdrlen+TCP_PKT);
	if(*bpp == nil)


@@ 964,25 972,25 @@ seq_within(ulong x, ulong low, ulong high)
int
seq_lt(ulong x, ulong y)
{
	return x < y;
	return (int)(x-y) < 0;
}

int
seq_le(ulong x, ulong y)
{
	return x <= y;
	return (int)(x-y) <= 0;
}

int
seq_gt(ulong x, ulong y)
{
	return x > y;
	return (int)(x-y) > 0;
}

int
seq_ge(ulong x, ulong y)
{
	return x >= y;
	return (int)(x-y) >= 0;
}

/*


@@ 1023,6 1031,28 @@ update(Conv *s, Tcp *seg)
		return;
	}

	/* added by Dong for fast retransmission */
	if( seg->ack == tcb->snd.una &&
	    seg->len == 0 && seg->wnd == tcb->snd.wnd ) {

		/* this is a pure ack w/o window update */
//		print("dupack %lud ack %lud sndwnd %d advwin %d\n",
//		tcb->snd.dupacks, seg->ack, tcb->snd.wnd, seg->wnd);

		if(++tcb->snd.dupacks == TCPREXMTTHRESH) {
			/*
			 *  tahoe tcp rxt the packet, half sshthresh,
 			 *  and set cwnd to one packet
			 */
			tcb->snd.recovery = 1;
			tcb->snd.rxt = tcb->snd.nxt;
//			print("fast rxt %lud, nxt %lud\n", tcb->snd.una, tcb->snd.nxt);
			tcprxmit(s, 0);
		} else {
			/* do reno tcp here. */
		}
	}

	if(seq_ge(seg->ack,tcb->snd.wl2))
	if(seq_gt(seg->seq,tcb->snd.wl1) || (seg->seq == tcb->snd.wl1)) {
		if(seg->wnd != 0 && tcb->snd.wnd == 0)


@@ 1036,12 1066,30 @@ update(Conv *s, Tcp *seg)
	if(!seq_gt(seg->ack, tcb->snd.una))
		return;

	/* something new was acked in this packet */
	tcb->flags |= ACKED;
        /*
	 *  any positive ack turns off fast rxt,
         *  (should we do new-reno on partial acks?)
	 */
	if(!tcb->snd.recovery || seq_ge(seg->ack, tcb->snd.rxt)) {
		tcb->snd.dupacks = 0;
		tcb->snd.recovery = 0;
	} else {
//		print("rxt next %lud, cwin %ud\n", seg->ack, tcb->cwind);
	}

	/* Compute the new send window size */
	acked = seg->ack - tcb->snd.una;
	if(tcb->cwind < tcb->snd.wnd) {

	/* avoid slow start and timers for SYN acks */
	if((tcb->flags & SYNACK) == 0) {
		tcb->flags |= SYNACK;
		acked--;
		tcb->sndcnt--;
		goto done;
	}

	/* slow start as long as we're not recovering from lost packets */
	if(tcb->cwind < tcb->snd.wnd && !tcb->snd.recovery) {
		if(tcb->cwind < tcb->ssthresh) {
			expand = tcb->mss;
			if(acked < expand)


@@ 1084,12 1132,7 @@ update(Conv *s, Tcp *seg)
		}
	}

	if((tcb->flags & SYNACK) == 0) {
		tcb->flags |= SYNACK;
		acked--;
		tcb->sndcnt--;
	}

done:
	qdiscard(s->wq, acked);

	tcb->sndcnt -= acked;


@@ 1539,7 1582,7 @@ tcpoutput(Conv *s)
	int msgs;
	Tcpctl *tcb;
	Block *hbp, *bp;
	int sndcnt, n, first;
	int sndcnt, n;
	ulong ssize, dsize, usable, sent;
	Fs *f;
	Tcppriv *tpriv;


@@ 1562,7 1605,6 @@ tcpoutput(Conv *s)
		tcb->flags |= FORCE;
	}

	first = tcb->snd.ptr == tcb->snd.una;
	for(msgs = 0; msgs < 100; msgs++) {
		sndcnt = tcb->sndcnt;
		sent = tcb->snd.ptr - tcb->snd.una;


@@ 1587,22 1629,7 @@ tcpoutput(Conv *s)
			if(tcb->snd.wnd < usable)
				usable = tcb->snd.wnd;
			usable -= sent;
			/*
			 *  hold small pieces in the hopes that more will come along.
			 *  this is pessimal in synchronous communications so go ahead
			 *  and send if any of the following:
			 *   - there's no unacked packets outstanding
			 *   - we've forced to send anyways
			 *   - we've just gotten an ACK for a previous packet
			 *   - there's more than 5 bytes queued
			 */
			if(!first)
			if(!(tcb->flags&(FORCE|ACKED)))
			if((sndcnt-sent) < 5)
				usable = 0;
		}
		tcb->flags &= ~ACKED;

		ssize = sndcnt-sent;
		if(usable < ssize)
			ssize = usable;


@@ 1615,6 1642,19 @@ tcpoutput(Conv *s)
		if((tcb->flags&FORCE) == 0)
			break;

		/* avoid sending short packets unless... */
		if(dsize != 0) {
			/* ...we have a full segment */
			if(dsize != tcb->mss)
			/* ...the data was just queued */
			if((dsize + sent) != sndcnt)
			/* ...we're being forced */
			if(!(tcb->flags&FORCE))
			/* ...we have at least half a window's worth to send */
			if(dsize < tcb->snd.wnd/2 || tcb->snd.wnd == 0)
				return;
		}

		tcphalt(tpriv, &tcb->acktimer);

		tcb->flags &= ~FORCE;


@@ 1670,11 1710,12 @@ tcpoutput(Conv *s)
			seg.flags |= PSH;

		/* keep track of balance of resent data */
		if(tcb->snd.ptr < tcb->snd.nxt) {
		if(seq_lt(tcb->snd.ptr, tcb->snd.nxt)) {
			n = tcb->snd.nxt - tcb->snd.ptr;
			if(ssize < n)
				n = ssize;
			tcb->resent += n;
			tpriv->tstats.tcpRetransSegs++;
		}

		tcb->snd.ptr += ssize;


@@ 1712,8 1753,12 @@ tcpoutput(Conv *s)
			if(tcb->timer.state != TimerON)
				tcpgo(tpriv, &tcb->timer);

			/* If round trip timer isn't running, start it */
			if(tcb->rtt_timer.state != TimerON) {
			/*  If round trip timer isn't running, start it.
			 *  measure the longest packet only in case the
			 *  transmission time dominates RTT
			 */
			if(tcb->rtt_timer.state != TimerON)
			if(ssize == tcb->mss) {
				tcpgo(tpriv, &tcb->rtt_timer);
				tcb->rttseq = tcb->snd.ptr;
			}


@@ 1810,31 1855,36 @@ tcpstartka(Conv *s, char **f, int n)
}

void
tcprxmit(Conv *s)
tcprxmit(Conv *s, int dolock)
{
	Tcpctl *tcb;
	Tcppriv *tpriv;

	tpriv = s->p->priv;
	tcb = (Tcpctl*)s->ptcl;

	qlock(s);
	if(dolock)
		qlock(s);
	tcb->flags |= RETRAN|FORCE;
	tcb->snd.ptr = tcb->snd.una;

	/* Pull window down to a single packet and halve the slow
	 * start threshold
	/*
	 *  We should be halving the slow start thershhold (down to one
	 *  mss) but leaving it at mss seems to work well enough
	 */
	tcb->ssthresh = tcb->cwind / 2;
	tcb->ssthresh = tcb->ssthresh;
	if(tcb->mss > tcb->ssthresh)
		tcb->ssthresh = tcb->mss;
//	win = (tcb->cwind<tcb->snd.wnd)?tcb->cwind:tcb->snd.wnd/ tcb->mss;
//	win = win/2;
//	if ( win < 2 )
//		win = 2;
//	tcb->ssthresh = win * tcb->mss;
 	tcb->ssthresh = tcb->mss;

	/*
	 *  pull window down to a single packet
	 */
	tcb->cwind = tcb->mss;
	tcpoutput(s);

	tpriv->tstats.tcpRetransSegs++;
	qunlock(s);
	if(dolock)
		qunlock(s);
}

void


@@ 1843,8 1893,10 @@ tcptimeout(void *arg)
	Conv *s;
	Tcpctl *tcb;
	int maxback;
	Tcppriv *tpriv;

	s = (Conv*)arg;
	tpriv = s->p->priv;
	tcb = (Tcpctl*)s->ptcl;




@@ 1859,7 1911,9 @@ tcptimeout(void *arg)
			localclose(s, Etimedout);
			break;
		}
		tcprxmit(s);
		tcprxmit(s, 1);
		tpriv->tstats.tcpRetransTimeouts++;
		tcb->snd.dupacks = 0;
		break;
	case Time_wait:
		localclose(s, nil);


@@ 2088,7 2142,7 @@ tcpstats(Proto *tcp, char *buf, int len)



	return snprint(buf, len, "%lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud",
	return snprint(buf, len, "%lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud %lud",
		tpriv->tstats.tcpRtoAlgorithm,
		tpriv->tstats.tcpRtoMin,
		tpriv->tstats.tcpRtoMax,


@@ 2101,6 2155,7 @@ tcpstats(Proto *tcp, char *buf, int len)
		tpriv->tstats.tcpInSegs,
		tpriv->tstats.tcpOutSegs,
		tpriv->tstats.tcpRetransSegs,
		tpriv->tstats.tcpRetransTimeouts,
		tpriv->tstats.InErrs,
		tpriv->tstats.OutRsts);
}