diff --git a/README.md b/README.md index ce447fa..9bfb5ed 100644 --- a/README.md +++ b/README.md @@ -67,6 +67,37 @@ go build -o /usr/local/bin/tsproxy . Target can be any MagicDNS name, short hostname, or tailnet IP. +### Prefer a tailnet IP for the target + +The dial path resolves a target in two stages. A literal IP is used as-is with +no lookup at all. A *name* is checked against the in-memory tailnet map, and on +a miss falls through to a real **system DNS** query — which can stall for +seconds on a short hostname, entirely on the local machine, before any packet +goes near the target. + +That shows up as forwarded dials alternating between milliseconds (resolver +warm) and seconds (resolver cold), which is long enough for a client to hit its +own timeout while it sits connected and unserved. If you see `SLOW DIAL` in the +logs, try the tailnet IP (`100.x.y.z:PORT`) or the fully-qualified MagicDNS name +instead of the short one. + +### Tuning for a low-latency link + +The defaults are conservative because a tailnet dial may need to renegotiate a +path or fall back to DERP. If your target is a millisecond away, much tighter +values are reasonable and will surface a dead target in about a second: + +```sh +tsproxy -f 127.0.0.1:8080=100.x.y.z:80 --dial-timeout 2s --probe-timeout 1s +``` + +**Fix resolution before tightening these.** A slow dial caused by system DNS is +not a slow link, and a timeout short enough to cut it off turns connections that +would have worked into failures — each one resetting a client and closing the +port until the next probe, which then pays the same DNS cost. Confirm dials are +consistently fast (`--trace` shows `connected to target in ...`) and only then +lower the budgets. + ## Running under launchd (macOS) An example LaunchAgent plist is included as diff --git a/main.go b/main.go index e057b98..407f69a 100644 --- a/main.go +++ b/main.go @@ -477,6 +477,12 @@ func (t *targetConn) failure() error { return t.err } +// A dial slower than this is reported without --trace. It is not a fraction of +// --dial-timeout on purpose: the budget is about when to give up, this is about +// how long a client will sit connected but unserved, which is a much smaller +// number and is what actually breaks applications. +const slowDial = time.Second + // which side of a forwarded connection finished first, and what it moved. type direction struct { name string @@ -566,10 +572,13 @@ func (p *proxy) handle(ctx context.Context, in net.Conn, id uint64) { p.suspend() return } - if dialTook > p.dialWait/4 { - // Worth seeing without --trace: a dial this slow is on its way to - // becoming a failure, and says the tailnet path is being rebuilt. - p.logf("[conn %d] slow dial: connected in %s (budget %s)", id, dialTook, p.dialWait) + if dialTook > slowDial { + // Worth seeing without --trace. This is measured against what a client + // will sit through, not against the dial budget: the client is already + // connected -- the local handshake completed the moment it called + // connect() -- so every second spent here is a second it waits with no + // idea anything is wrong, and long enough will trip its own timeout. + p.logf("[conn %d] SLOW DIAL: connected in %s -- the client has been waiting that long", id, dialTook) } else { p.tracef("[conn %d] connected to target in %s", id, dialTook) } @@ -612,7 +621,11 @@ func (p *proxy) handle(ctx context.Context, in net.Conn, id uint64) { abort(in) p.suspend() case first.err != nil: - p.tracef("[conn %d] reset after %s: %s: %v (%d bytes)", id, lived, first.name, first.err, first.bytes) + // The target is wired through targetConn, so any failure of its own + // would have set targetErr and taken the branch above. Reaching here + // means the local client is what broke -- it gave up, or went away. + p.logf("[conn %d] client gave up after %s: %v (%s moved %d bytes)", + id, lived, first.err, first.name, first.bytes) abort(in) default: p.tracef("[conn %d] closed cleanly after %s: %s ended, %d bytes", id, lived, first.name, first.bytes)