mirror of
https://github.com/MHSanaei/3x-ui.git
synced 2026-08-27 03:44:23 +00:00
fix(tunnelmonitor): observable recovery, signal headroom, and hardening
Address the remaining review findings on the tunnel health monitor: - Recovery is now synchronous and observable: the callback calls server.RestartXray() directly and returns its error instead of just enqueuing SIGUSR1, so a failed restart no longer masks as success and arms the cooldown while the tunnel is still down. - Give the OS signal channel headroom (buffer 8) so producers cannot starve a SIGTERM/SIGINT out of the single slot. - Warn at startup when the monitor is enabled without a proxy, since the probe then measures host connectivity rather than the xray tunnel. - Cap failures at the threshold in the nil-recover branch too, matching the cooldown cap. - Document the XUI_TUNNEL_HEALTH_* vars in .env.example and the README. - Add tests for status-code classification, Normalize bounds, New proxy scheme errors, the recovery-error and nil-recover paths, the cooldown cap, and Run context cancellation (coverage 90%).
This commit is contained in:
parent
768ba1a6f5
commit
c268b91ef8
5 changed files with 279 additions and 8 deletions
13
.env.example
13
.env.example
|
|
@ -4,3 +4,16 @@ XUI_LOG_FOLDER=x-ui
|
|||
XUI_BIN_FOLDER=x-ui
|
||||
XUI_INIT_WEB_BASE_PATH=/
|
||||
# XUI_PORT=8080
|
||||
|
||||
# Optional tunnel health monitor (disabled by default). It periodically probes a
|
||||
# URL and restarts xray-core after repeated failures. Point XUI_TUNNEL_HEALTH_PROXY
|
||||
# at a local xray inbound so the probe tests the tunnel; without it the probe only
|
||||
# checks host connectivity and a restart will not fix host network issues. A restart
|
||||
# drops every connected client.
|
||||
# XUI_TUNNEL_HEALTH_MONITOR=true
|
||||
# XUI_TUNNEL_HEALTH_PROXY=socks5://127.0.0.1:1080
|
||||
# XUI_TUNNEL_HEALTH_URL=https://www.cloudflare.com/cdn-cgi/trace
|
||||
# XUI_TUNNEL_HEALTH_INTERVAL=30s
|
||||
# XUI_TUNNEL_HEALTH_TIMEOUT=10s
|
||||
# XUI_TUNNEL_HEALTH_FAILURES=3
|
||||
# XUI_TUNNEL_HEALTH_COOLDOWN=5m
|
||||
|
|
|
|||
|
|
@ -146,6 +146,13 @@ docker run -d --cap-add=NET_ADMIN --cap-add=NET_RAW ... ghcr.io/mhsanaei/3x-ui
|
|||
| `XUI_ENABLE_FAIL2BAN` | Enable Fail2ban-based IP-limit enforcement | `true` |
|
||||
| `XUI_LOG_LEVEL` | Log verbosity (`debug`, `info`, `warning`, `error`) | `info` |
|
||||
| `XUI_DEBUG` | Enable debug mode | `false` |
|
||||
| `XUI_TUNNEL_HEALTH_MONITOR` | Enable the tunnel health monitor (probes a URL and restarts xray after repeated failures; a restart drops all clients) | `false` |
|
||||
| `XUI_TUNNEL_HEALTH_PROXY` | Proxy the probe is sent through; point it at a local xray inbound so the probe tests the tunnel (e.g. `socks5://127.0.0.1:1080`). Empty means the probe only checks host connectivity | — |
|
||||
| `XUI_TUNNEL_HEALTH_URL` | URL probed for tunnel health | `https://www.cloudflare.com/cdn-cgi/trace` |
|
||||
| `XUI_TUNNEL_HEALTH_INTERVAL` | Interval between probes | `30s` |
|
||||
| `XUI_TUNNEL_HEALTH_TIMEOUT` | Per-probe timeout | `10s` |
|
||||
| `XUI_TUNNEL_HEALTH_FAILURES` | Consecutive failures before a restart is triggered | `3` |
|
||||
| `XUI_TUNNEL_HEALTH_COOLDOWN` | Minimum delay between consecutive restarts | `5m` |
|
||||
|
||||
## Supported Languages
|
||||
|
||||
|
|
|
|||
|
|
@ -184,6 +184,7 @@ func (m *Monitor) Step(ctx context.Context) (bool, error) {
|
|||
}
|
||||
|
||||
if m.recover == nil {
|
||||
m.failures = m.cfg.FailureThreshold
|
||||
return false, errors.New("recovery function is not configured")
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ import (
|
|||
"errors"
|
||||
"net/http"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
|
|
@ -202,3 +203,252 @@ func TestConfigFromEnvParsesValues(t *testing.T) {
|
|||
t.Fatalf("unexpected cooldown: %s", cfg.Cooldown)
|
||||
}
|
||||
}
|
||||
|
||||
func failingClient() *http.Client {
|
||||
return &http.Client{
|
||||
Transport: roundTripFunc(func(req *http.Request) (*http.Response, error) {
|
||||
return nil, errors.New("tunnel down")
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
func statusClient(code int) *http.Client {
|
||||
return &http.Client{
|
||||
Transport: roundTripFunc(func(req *http.Request) (*http.Response, error) {
|
||||
return &http.Response{StatusCode: code, Body: http.NoBody}, nil
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
func TestProbeStatusCodeClassification(t *testing.T) {
|
||||
cases := []struct {
|
||||
status int
|
||||
healthy bool
|
||||
}{
|
||||
{199, false},
|
||||
{200, true},
|
||||
{204, true},
|
||||
{301, true},
|
||||
{399, true},
|
||||
{400, false},
|
||||
{404, false},
|
||||
{500, false},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
cfg := Config{
|
||||
Enabled: true,
|
||||
URL: "http://example.test",
|
||||
Interval: time.Minute,
|
||||
Timeout: time.Second,
|
||||
FailureThreshold: 100,
|
||||
Cooldown: time.Minute,
|
||||
}
|
||||
|
||||
monitor := newWithClient(cfg, statusClient(tc.status), func(ctx context.Context) error {
|
||||
return nil
|
||||
})
|
||||
|
||||
recovered, err := monitor.Step(context.Background())
|
||||
if recovered {
|
||||
t.Fatalf("status %d: unexpected recovery", tc.status)
|
||||
}
|
||||
if tc.healthy && err != nil {
|
||||
t.Fatalf("status %d: expected healthy probe, got error %v", tc.status, err)
|
||||
}
|
||||
if !tc.healthy && err == nil {
|
||||
t.Fatalf("status %d: expected failure, got nil error", tc.status)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeClampsBounds(t *testing.T) {
|
||||
got := Config{
|
||||
URL: " ",
|
||||
Interval: 0,
|
||||
Timeout: 500 * time.Millisecond,
|
||||
FailureThreshold: 0,
|
||||
Cooldown: 0,
|
||||
}.Normalize()
|
||||
|
||||
if got.URL != defaultHealthURL {
|
||||
t.Fatalf("URL not defaulted: %q", got.URL)
|
||||
}
|
||||
if got.Interval != defaultInterval {
|
||||
t.Fatalf("Interval not clamped: %s", got.Interval)
|
||||
}
|
||||
if got.Timeout != defaultTimeout {
|
||||
t.Fatalf("Timeout not clamped: %s", got.Timeout)
|
||||
}
|
||||
if got.FailureThreshold != defaultFailureThreshold {
|
||||
t.Fatalf("FailureThreshold not clamped: %d", got.FailureThreshold)
|
||||
}
|
||||
if got.Cooldown != defaultCooldown {
|
||||
t.Fatalf("Cooldown not clamped: %s", got.Cooldown)
|
||||
}
|
||||
|
||||
valid := Config{
|
||||
URL: "https://example.com/health",
|
||||
Interval: 15 * time.Second,
|
||||
Timeout: 3 * time.Second,
|
||||
FailureThreshold: 5,
|
||||
Cooldown: 2 * time.Minute,
|
||||
}.Normalize()
|
||||
|
||||
if valid.URL != "https://example.com/health" ||
|
||||
valid.Interval != 15*time.Second ||
|
||||
valid.Timeout != 3*time.Second ||
|
||||
valid.FailureThreshold != 5 ||
|
||||
valid.Cooldown != 2*time.Minute {
|
||||
t.Fatalf("valid config was mutated: %+v", valid)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRejectsUnsupportedProxyScheme(t *testing.T) {
|
||||
m, err := New(Config{ProxyURL: "ftp://127.0.0.1:21"}, func(ctx context.Context) error {
|
||||
return nil
|
||||
})
|
||||
if err == nil || m != nil {
|
||||
t.Fatalf("expected error and nil monitor for bad scheme, got m=%v err=%v", m, err)
|
||||
}
|
||||
|
||||
m, err = New(Config{}, func(ctx context.Context) error {
|
||||
return nil
|
||||
})
|
||||
if err != nil || m == nil {
|
||||
t.Fatalf("expected a valid monitor for empty proxy, got m=%v err=%v", m, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorRecoveryErrorDoesNotArmCooldown(t *testing.T) {
|
||||
cfg := Config{
|
||||
Enabled: true,
|
||||
URL: "http://example.test",
|
||||
Interval: time.Minute,
|
||||
Timeout: time.Second,
|
||||
FailureThreshold: 1,
|
||||
Cooldown: time.Minute,
|
||||
}
|
||||
|
||||
attempts := 0
|
||||
monitor := newWithClient(cfg, failingClient(), func(ctx context.Context) error {
|
||||
attempts++
|
||||
return errors.New("restart failed")
|
||||
})
|
||||
monitor.now = func() time.Time {
|
||||
return time.Unix(100, 0)
|
||||
}
|
||||
|
||||
recovered, err := monitor.Step(context.Background())
|
||||
if recovered || err == nil {
|
||||
t.Fatalf("failed recovery must report recovered=false with an error, got recovered=%v err=%v", recovered, err)
|
||||
}
|
||||
if !monitor.lastRecovery.IsZero() {
|
||||
t.Fatal("a failed recovery must not arm the cooldown")
|
||||
}
|
||||
|
||||
if _, err := monitor.Step(context.Background()); err == nil {
|
||||
t.Fatal("expected error on the second failing step")
|
||||
}
|
||||
if attempts != 2 {
|
||||
t.Fatalf("recovery should be retried (no cooldown) after a failure, attempts=%d", attempts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorNilRecoverStaysBounded(t *testing.T) {
|
||||
cfg := Config{
|
||||
Enabled: true,
|
||||
URL: "http://example.test",
|
||||
Interval: time.Minute,
|
||||
Timeout: time.Second,
|
||||
FailureThreshold: 2,
|
||||
Cooldown: time.Minute,
|
||||
}
|
||||
|
||||
monitor := newWithClient(cfg, failingClient(), nil)
|
||||
|
||||
for i := 0; i < 5; i++ {
|
||||
recovered, _ := monitor.Step(context.Background())
|
||||
if recovered {
|
||||
t.Fatal("a nil recovery func must never report recovery")
|
||||
}
|
||||
if monitor.failures > cfg.FailureThreshold {
|
||||
t.Fatalf("failures must stay capped at threshold %d, got %d", cfg.FailureThreshold, monitor.failures)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorFailuresCappedDuringCooldown(t *testing.T) {
|
||||
cfg := Config{
|
||||
Enabled: true,
|
||||
URL: "http://example.test",
|
||||
Interval: time.Minute,
|
||||
Timeout: time.Second,
|
||||
FailureThreshold: 2,
|
||||
Cooldown: time.Minute,
|
||||
}
|
||||
|
||||
restarts := 0
|
||||
monitor := newWithClient(cfg, failingClient(), func(ctx context.Context) error {
|
||||
restarts++
|
||||
return nil
|
||||
})
|
||||
monitor.now = func() time.Time {
|
||||
return time.Unix(100, 0)
|
||||
}
|
||||
|
||||
monitor.Step(context.Background())
|
||||
if recovered, _ := monitor.Step(context.Background()); !recovered {
|
||||
t.Fatal("expected recovery once the threshold is reached")
|
||||
}
|
||||
|
||||
for i := 0; i < 6; i++ {
|
||||
monitor.Step(context.Background())
|
||||
if monitor.failures > cfg.FailureThreshold {
|
||||
t.Fatalf("failures must never exceed threshold %d during cooldown, got %d", cfg.FailureThreshold, monitor.failures)
|
||||
}
|
||||
}
|
||||
|
||||
if restarts != 1 {
|
||||
t.Fatalf("cooldown should suppress further recoveries, restarts=%d", restarts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMonitorRunStopsOnContextCancel(t *testing.T) {
|
||||
cfg := Config{
|
||||
Enabled: true,
|
||||
URL: "http://example.test",
|
||||
Timeout: time.Second,
|
||||
FailureThreshold: 1,
|
||||
Cooldown: time.Hour,
|
||||
}
|
||||
|
||||
recovered := make(chan struct{})
|
||||
var once sync.Once
|
||||
monitor := newWithClient(cfg, failingClient(), func(ctx context.Context) error {
|
||||
once.Do(func() { close(recovered) })
|
||||
return nil
|
||||
})
|
||||
monitor.cfg.Interval = 5 * time.Millisecond
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
monitor.Run(ctx)
|
||||
close(done)
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-recovered:
|
||||
case <-time.After(2 * time.Second):
|
||||
cancel()
|
||||
t.Fatal("Run did not trigger recovery within the deadline")
|
||||
}
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("Run did not return after context cancellation")
|
||||
}
|
||||
}
|
||||
|
|
|
|||
16
main.go
16
main.go
|
|
@ -93,7 +93,7 @@ func runWebServer() {
|
|||
return
|
||||
}
|
||||
|
||||
sigCh := make(chan os.Signal, 1)
|
||||
sigCh := make(chan os.Signal, 8)
|
||||
// Trap shutdown signals
|
||||
signal.Notify(sigCh, syscall.SIGHUP, syscall.SIGTERM, sys.SIGUSR1, os.Interrupt)
|
||||
global.SetRestartHook(func() {
|
||||
|
|
@ -106,16 +106,16 @@ func runWebServer() {
|
|||
var stopTunnelHealthMonitor context.CancelFunc
|
||||
monitorCfg := tunnelmonitor.ConfigFromEnv()
|
||||
if monitorCfg.Enabled {
|
||||
if monitorCfg.ProxyURL == "" {
|
||||
logger.Warning("Tunnel health monitor enabled without XUI_TUNNEL_HEALTH_PROXY: the probe measures host connectivity, not the xray tunnel, so failures will restart xray without fixing host network issues")
|
||||
}
|
||||
|
||||
monitorCtx, cancel := context.WithCancel(context.Background())
|
||||
stopTunnelHealthMonitor = cancel
|
||||
|
||||
monitor, err := tunnelmonitor.New(monitorCfg, func(ctx context.Context) error {
|
||||
select {
|
||||
case sigCh <- sys.SIGUSR1:
|
||||
return nil
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
}
|
||||
monitor, err := tunnelmonitor.New(monitorCfg, func(_ context.Context) error {
|
||||
logger.Warning("Tunnel health monitor threshold reached, restarting xray-core")
|
||||
return server.RestartXray()
|
||||
})
|
||||
if err != nil {
|
||||
logger.Warning("Tunnel health monitor disabled: ", err)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue