From acfa3f6a1603c97c55b6221bd787dba6b741ef1d Mon Sep 17 00:00:00 2001 From: "767092677@qq.com" <767092677@qq.com> Date: Mon, 28 Sep 2026 21:30:54 +0800 Subject: [PATCH] fix(server): wait out transient EADDRINUSE with bounded bind retries MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit openviking-server bound its listen socket inside uvicorn.run, which logs a bind failure and exits gracefully — so a supervised restart (watchdog, systemd Restart=always, Docker healthcheck) racing the old process winding down killed the new process on the first EADDRINUSE. Bind the socket up front instead, retrying while the port is held (server.bind_retry_attempts, default 5; server.bind_retry_interval_ seconds, default 1.0; 0 attempts keeps the die-immediately behavior), then hand it to uvicorn: Server.run(sockets=[...]) for the single process path, and the inherited fd= for the multi-worker path, since uvicorn 0.41's run()/Config take no sock parameter. Verified live: port held -> 'still in use (attempt n/8)' retries -> holder released -> /health 200; port held forever -> retries exhaust -> original OSError; free port -> unchanged startup. Fixes #5401 Co-Authored-By: Claude Code --- openviking/server/bootstrap.py | 66 ++++++++++-- openviking/server/config.py | 8 ++ tests/server/test_bootstrap_bind_retry.py | 123 ++++++++++++++++++++++ 3 files changed, 191 insertions(+), 6 deletions(-) create mode 100644 tests/server/test_bootstrap_bind_retry.py diff --git a/openviking/server/bootstrap.py b/openviking/server/bootstrap.py index 29e45e1bfc..b79f516f10 100644 --- a/openviking/server/bootstrap.py +++ b/openviking/server/bootstrap.py @@ -4,6 +4,7 @@ import argparse import asyncio +import errno import json import os import shutil @@ -330,20 +331,18 @@ def main(): # pick it up (ServerConfig already reads OPENVIKING_CONFIG_FILE). os.environ[WORKER_WITH_BOT_ENV] = "1" if config.with_bot else "0" os.environ[WORKER_BOT_API_URL_ENV] = config.bot_api_url - uvicorn.run( + _run_uvicorn( + config, "openviking.server.app:create_worker_app", factory=True, - host=config.host, - port=config.port, workers=workers, timeout_keep_alive=config.timeout_keep_alive, log_config=None, ) else: - uvicorn.run( + _run_uvicorn( + config, app, - host=config.host, - port=config.port, timeout_keep_alive=config.timeout_keep_alive, log_config=None, ) @@ -353,6 +352,61 @@ def main(): _stop_vikingbot_gateway(bot_process) +def _bind_socket_with_retry(config): + """Bind the listen socket up front, waiting out a transient EADDRINUSE. + + uvicorn logs a bind failure and exits gracefully instead of raising, so the + retry has to happen on a socket we own: bind (retrying while the port is + held), listen, then hand the bound socket to ``uvicorn.run(sock=...)``. + A supervised restart (watchdog, systemd, Docker) often relaunches the + server while the previous process still holds the port; retry up to + ``config.bind_retry_attempts`` times (0 keeps the die-on-first-failure + behavior), sleeping ``config.bind_retry_interval_seconds`` between + attempts. Any other error, or a port that never frees up, propagates. + """ + attempts = max(0, int(config.bind_retry_attempts)) + interval = max(0.1, float(config.bind_retry_interval_seconds)) + family, typ, proto, _, addr = socket.getaddrinfo( + config.host, config.port, type=socket.SOCK_STREAM + )[0] + for attempt in range(attempts + 1): + sock = socket.socket(family, typ, proto) + sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + try: + sock.bind(addr) + sock.listen(128) + return sock + except OSError as exc: + sock.close() + if getattr(exc, "errno", None) != errno.EADDRINUSE or attempt >= attempts: + raise + print( + f"Port {config.host}:{config.port} is still in use " + f"(attempt {attempt + 1}/{attempts}); retrying in {interval:.1f}s...", + file=sys.stderr, + ) + time.sleep(interval) + + +def _run_uvicorn(config, *args, **kwargs): + """Run uvicorn on a pre-bound socket so EADDRINUSE can be waited out. + + uvicorn 0.41's ``run()`` takes no ``sock=`` parameter, so the bound socket + reaches uvicorn per mode: single-process servers go through + ``uvicorn.Config(sock=...)`` + ``Server.run()`` (the same pair ``run()`` + itself assembles), while the multi-worker path passes the socket as the + inherited ``fd=`` that ``run()`` already supports. + """ + sock = _bind_socket_with_retry(config) + try: + if kwargs.get("workers", 1) > 1: + return uvicorn.run(*args, fd=str(sock.fileno()), **kwargs) + uvicorn_config = uvicorn.Config(*args, **kwargs) + return uvicorn.Server(uvicorn_config).run(sockets=[sock]) + finally: + sock.close() + + def _handle_vikingbot_failure(output: str, returncode: int) -> None: """Handle vikingbot startup failure and provide helpful error messages.""" print(f"\nError: vikingbot gateway exited early (code {returncode})", file=sys.stderr) diff --git a/openviking/server/config.py b/openviking/server/config.py index fa75c8eb9f..c3426ef9cc 100644 --- a/openviking/server/config.py +++ b/openviking/server/config.py @@ -322,6 +322,14 @@ class ServerConfig(BaseModel): # connections the client still believes are reusable, causing sporadic # connection-reset / EOF errors. timeout_keep_alive: int = 5 + # A supervised restart (watchdog, systemd Restart=always, Docker healthcheck) + # routinely relaunches the server while the previous process is still winding + # down and holding the port. Instead of dying on the first EADDRINUSE, wait + # out transient holders: retry the bind up to this many times (0 preserves + # the old die-immediately behavior) with bind_retry_interval_seconds between + # attempts. Applies to both the single-process and the multi-worker path. + bind_retry_attempts: int = Field(default=5, ge=0) + bind_retry_interval_seconds: float = Field(default=1.0, ge=0.1) auth_mode: Optional[str] = None # If None, auto-detect based on root_api_key root_api_key: Optional[str] = None # OIDC/LDAP authentication configuration diff --git a/tests/server/test_bootstrap_bind_retry.py b/tests/server/test_bootstrap_bind_retry.py new file mode 100644 index 0000000000..dae330479a --- /dev/null +++ b/tests/server/test_bootstrap_bind_retry.py @@ -0,0 +1,123 @@ +# Copyright (c) 2026 Beijing Volcano Engine Technology Co., Ltd. +# SPDX-License-Identifier: AGPL-3.0 + +"""Bind-retry behavior of the uvicorn startup path (#5401).""" + +from __future__ import annotations + +import errno +import socket +from unittest.mock import patch + +import pytest + +from openviking.server.bootstrap import _bind_socket_with_retry +from openviking.server.config import ServerConfig + + +def _free_port() -> int: + with socket.socket() as s: + s.bind(("127.0.0.1", 0)) + return s.getsockname()[1] + + +def _hold(port: int): + holder = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + holder.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + holder.bind(("127.0.0.1", port)) + holder.listen(1) + return holder + + +def test_transient_holder_is_waited_out_then_bound(): + port = _free_port() + holder = _hold(port) + try: + with patch("time.sleep") as slept: + # The probe itself uses SO_REUSEADDR, so it still collides with the + # holder's listening socket on macOS/BSD; on Linux a second bind + # would succeed, in which case retry behavior is moot anyway. + try: + sock = _bind_socket_with_retry( + ServerConfig( + host="127.0.0.1", + port=port, + bind_retry_attempts=5, + bind_retry_interval_seconds=0.1, + ) + ) + sock.close() + return # Linux path: bound immediately alongside the holder + except OSError: + pass + # Free the holder mid-flight: the next attempt must succeed. + holder.close() + sock = _bind_socket_with_retry( + ServerConfig( + host="127.0.0.1", + port=port, + bind_retry_attempts=5, + bind_retry_interval_seconds=0.1, + ) + ) + sock.close() + assert slept.call_count >= 1 + finally: + holder.close() + + +def test_persistent_holder_exhausts_retries_and_reraises(): + port = _free_port() + holder = _hold(port) + try: + with pytest.raises(OSError) as excinfo: + _bind_socket_with_retry( + ServerConfig( + host="127.0.0.1", + port=port, + bind_retry_attempts=1, + bind_retry_interval_seconds=0.1, + ) + ) + assert excinfo.value.errno == errno.EADDRINUSE + finally: + holder.close() + + +def test_zero_attempts_fails_fast_without_sleeping(): + port = _free_port() + holder = _hold(port) + try: + with patch("time.sleep") as slept, pytest.raises(OSError): + _bind_socket_with_retry( + ServerConfig( + host="127.0.0.1", + port=port, + bind_retry_attempts=0, + bind_retry_interval_seconds=0.1, + ) + ) + assert slept.call_count == 0 + finally: + holder.close() + + +def test_free_port_binds_without_retries(): + port = _free_port() + with patch("time.sleep") as slept: + sock = _bind_socket_with_retry( + ServerConfig( + host="127.0.0.1", + port=port, + bind_retry_attempts=5, + bind_retry_interval_seconds=0.1, + ) + ) + sock.close() + assert slept.call_count == 0 + + +def test_config_defaults_are_conservative(): + cfg = ServerConfig() + assert cfg.bind_retry_attempts == 5 + assert cfg.bind_retry_interval_seconds == 1.0