more rst fixes 2
This commit is contained in:
parent
84ed794f5c
commit
1e59753238
2 changed files with 49 additions and 5 deletions
31
README.md
31
README.md
|
|
@ -67,6 +67,37 @@ go build -o /usr/local/bin/tsproxy .
|
|||
|
||||
Target can be any MagicDNS name, short hostname, or tailnet IP.
|
||||
|
||||
### Prefer a tailnet IP for the target
|
||||
|
||||
The dial path resolves a target in two stages. A literal IP is used as-is with
|
||||
no lookup at all. A *name* is checked against the in-memory tailnet map, and on
|
||||
a miss falls through to a real **system DNS** query — which can stall for
|
||||
seconds on a short hostname, entirely on the local machine, before any packet
|
||||
goes near the target.
|
||||
|
||||
That shows up as forwarded dials alternating between milliseconds (resolver
|
||||
warm) and seconds (resolver cold), which is long enough for a client to hit its
|
||||
own timeout while it sits connected and unserved. If you see `SLOW DIAL` in the
|
||||
logs, try the tailnet IP (`100.x.y.z:PORT`) or the fully-qualified MagicDNS name
|
||||
instead of the short one.
|
||||
|
||||
### Tuning for a low-latency link
|
||||
|
||||
The defaults are conservative because a tailnet dial may need to renegotiate a
|
||||
path or fall back to DERP. If your target is a millisecond away, much tighter
|
||||
values are reasonable and will surface a dead target in about a second:
|
||||
|
||||
```sh
|
||||
tsproxy -f 127.0.0.1:8080=100.x.y.z:80 --dial-timeout 2s --probe-timeout 1s
|
||||
```
|
||||
|
||||
**Fix resolution before tightening these.** A slow dial caused by system DNS is
|
||||
not a slow link, and a timeout short enough to cut it off turns connections that
|
||||
would have worked into failures — each one resetting a client and closing the
|
||||
port until the next probe, which then pays the same DNS cost. Confirm dials are
|
||||
consistently fast (`--trace` shows `connected to target in ...`) and only then
|
||||
lower the budgets.
|
||||
|
||||
## Running under launchd (macOS)
|
||||
|
||||
An example LaunchAgent plist is included as
|
||||
|
|
|
|||
23
main.go
23
main.go
|
|
@ -477,6 +477,12 @@ func (t *targetConn) failure() error {
|
|||
return t.err
|
||||
}
|
||||
|
||||
// A dial slower than this is reported without --trace. It is not a fraction of
|
||||
// --dial-timeout on purpose: the budget is about when to give up, this is about
|
||||
// how long a client will sit connected but unserved, which is a much smaller
|
||||
// number and is what actually breaks applications.
|
||||
const slowDial = time.Second
|
||||
|
||||
// which side of a forwarded connection finished first, and what it moved.
|
||||
type direction struct {
|
||||
name string
|
||||
|
|
@ -566,10 +572,13 @@ func (p *proxy) handle(ctx context.Context, in net.Conn, id uint64) {
|
|||
p.suspend()
|
||||
return
|
||||
}
|
||||
if dialTook > p.dialWait/4 {
|
||||
// Worth seeing without --trace: a dial this slow is on its way to
|
||||
// becoming a failure, and says the tailnet path is being rebuilt.
|
||||
p.logf("[conn %d] slow dial: connected in %s (budget %s)", id, dialTook, p.dialWait)
|
||||
if dialTook > slowDial {
|
||||
// Worth seeing without --trace. This is measured against what a client
|
||||
// will sit through, not against the dial budget: the client is already
|
||||
// connected -- the local handshake completed the moment it called
|
||||
// connect() -- so every second spent here is a second it waits with no
|
||||
// idea anything is wrong, and long enough will trip its own timeout.
|
||||
p.logf("[conn %d] SLOW DIAL: connected in %s -- the client has been waiting that long", id, dialTook)
|
||||
} else {
|
||||
p.tracef("[conn %d] connected to target in %s", id, dialTook)
|
||||
}
|
||||
|
|
@ -612,7 +621,11 @@ func (p *proxy) handle(ctx context.Context, in net.Conn, id uint64) {
|
|||
abort(in)
|
||||
p.suspend()
|
||||
case first.err != nil:
|
||||
p.tracef("[conn %d] reset after %s: %s: %v (%d bytes)", id, lived, first.name, first.err, first.bytes)
|
||||
// The target is wired through targetConn, so any failure of its own
|
||||
// would have set targetErr and taken the branch above. Reaching here
|
||||
// means the local client is what broke -- it gave up, or went away.
|
||||
p.logf("[conn %d] client gave up after %s: %v (%s moved %d bytes)",
|
||||
id, lived, first.err, first.name, first.bytes)
|
||||
abort(in)
|
||||
default:
|
||||
p.tracef("[conn %d] closed cleanly after %s: %s ended, %d bytes", id, lived, first.name, first.bytes)
|
||||
|
|
|
|||
Loading…
Reference in a new issue