more rst fixes 2

This commit is contained in:
Stefan Wasilewski 2026-08-01 05:37:46 +04:00
parent 84ed794f5c
commit 1e59753238
2 changed files with 49 additions and 5 deletions

View file

@ -67,6 +67,37 @@ go build -o /usr/local/bin/tsproxy .
Target can be any MagicDNS name, short hostname, or tailnet IP.
### Prefer a tailnet IP for the target
The dial path resolves a target in two stages. A literal IP is used as-is with
no lookup at all. A *name* is checked against the in-memory tailnet map, and on
a miss falls through to a real **system DNS** query — which can stall for
seconds on a short hostname, entirely on the local machine, before any packet
goes near the target.
That shows up as forwarded dials alternating between milliseconds (resolver
warm) and seconds (resolver cold), which is long enough for a client to hit its
own timeout while it sits connected and unserved. If you see `SLOW DIAL` in the
logs, try the tailnet IP (`100.x.y.z:PORT`) or the fully-qualified MagicDNS name
instead of the short one.
### Tuning for a low-latency link
The defaults are conservative because a tailnet dial may need to renegotiate a
path or fall back to DERP. If your target is a millisecond away, much tighter
values are reasonable and will surface a dead target in about a second:
```sh
tsproxy -f 127.0.0.1:8080=100.x.y.z:80 --dial-timeout 2s --probe-timeout 1s
```
**Fix resolution before tightening these.** A slow dial caused by system DNS is
not a slow link, and a timeout short enough to cut it off turns connections that
would have worked into failures — each one resetting a client and closing the
port until the next probe, which then pays the same DNS cost. Confirm dials are
consistently fast (`--trace` shows `connected to target in ...`) and only then
lower the budgets.
## Running under launchd (macOS)
An example LaunchAgent plist is included as

23
main.go
View file

@ -477,6 +477,12 @@ func (t *targetConn) failure() error {
return t.err
}
// A dial slower than this is reported without --trace. It is not a fraction of
// --dial-timeout on purpose: the budget is about when to give up, this is about
// how long a client will sit connected but unserved, which is a much smaller
// number and is what actually breaks applications.
const slowDial = time.Second
// which side of a forwarded connection finished first, and what it moved.
type direction struct {
name string
@ -566,10 +572,13 @@ func (p *proxy) handle(ctx context.Context, in net.Conn, id uint64) {
p.suspend()
return
}
if dialTook > p.dialWait/4 {
// Worth seeing without --trace: a dial this slow is on its way to
// becoming a failure, and says the tailnet path is being rebuilt.
p.logf("[conn %d] slow dial: connected in %s (budget %s)", id, dialTook, p.dialWait)
if dialTook > slowDial {
// Worth seeing without --trace. This is measured against what a client
// will sit through, not against the dial budget: the client is already
// connected -- the local handshake completed the moment it called
// connect() -- so every second spent here is a second it waits with no
// idea anything is wrong, and long enough will trip its own timeout.
p.logf("[conn %d] SLOW DIAL: connected in %s -- the client has been waiting that long", id, dialTook)
} else {
p.tracef("[conn %d] connected to target in %s", id, dialTook)
}
@ -612,7 +621,11 @@ func (p *proxy) handle(ctx context.Context, in net.Conn, id uint64) {
abort(in)
p.suspend()
case first.err != nil:
p.tracef("[conn %d] reset after %s: %s: %v (%d bytes)", id, lived, first.name, first.err, first.bytes)
// The target is wired through targetConn, so any failure of its own
// would have set targetErr and taken the branch above. Reaching here
// means the local client is what broke -- it gave up, or went away.
p.logf("[conn %d] client gave up after %s: %v (%s moved %d bytes)",
id, lived, first.err, first.name, first.bytes)
abort(in)
default:
p.tracef("[conn %d] closed cleanly after %s: %s ended, %d bytes", id, lived, first.name, first.bytes)