feat: multi-domain failover for provider accounts
Each ProviderAccount can now have multiple base URLs (provider_urls table). On stream failure, BroadcastGroup cycles to the next domain immediately with no wait; backs off only after all domains have been tried once. Background health checker pings every domain every 5 min via player_api.php and updates status/response_ms. Admin UI shows domain list with color-coded status badges and a "Verificar todos" button per provider.
This commit is contained in:
+56
-16
@@ -51,6 +51,14 @@ _SENTINEL = object() # signals end-of-stream to client queues
|
||||
_FFMPEG = shutil.which("ffmpeg") or "ffmpeg"
|
||||
|
||||
|
||||
def _url_netloc(url: str) -> str:
|
||||
from urllib.parse import urlparse
|
||||
try:
|
||||
return urlparse(url).netloc
|
||||
except Exception:
|
||||
return url
|
||||
|
||||
|
||||
@dataclass
|
||||
class ClientHandle:
|
||||
client_id: str
|
||||
@@ -75,14 +83,24 @@ class BroadcastGroup:
|
||||
def __init__(
|
||||
self,
|
||||
channel_id: int,
|
||||
stream_url: str,
|
||||
stream_urls: list[str],
|
||||
provider_account_id: int,
|
||||
channel_name: str = "",
|
||||
provider_name: str = "",
|
||||
# kept for backward compat in callers that still pass stream_url as kwarg
|
||||
stream_url: str | None = None,
|
||||
):
|
||||
self.channel_id = channel_id
|
||||
self.channel_name = channel_name
|
||||
self.stream_url = stream_url
|
||||
# Support old callers that pass a single stream_url
|
||||
if stream_urls:
|
||||
self._stream_urls = stream_urls
|
||||
elif stream_url:
|
||||
self._stream_urls = [stream_url]
|
||||
else:
|
||||
self._stream_urls = []
|
||||
self.stream_url = self._stream_urls[0] if self._stream_urls else ""
|
||||
self._active_url: str = self.stream_url
|
||||
self.provider_account_id = provider_account_id
|
||||
self.provider_name = provider_name
|
||||
self.started_at = time.time()
|
||||
@@ -212,6 +230,9 @@ class BroadcastGroup:
|
||||
# ffmpeg process info
|
||||
"ffmpeg_pid": self._ffmpeg_pid,
|
||||
"ffmpeg_cpu_pct": round(self._ffmpeg_cpu, 2),
|
||||
# multi-domain failover info
|
||||
"active_url_domain": _url_netloc(self._active_url),
|
||||
"url_count": len(self._stream_urls),
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
@@ -256,11 +277,16 @@ class BroadcastGroup:
|
||||
async def _pump_with_retry(self) -> None:
|
||||
consecutive = 0
|
||||
base_wait = 0.5
|
||||
url_idx = 0
|
||||
n = len(self._stream_urls) or 1
|
||||
|
||||
try:
|
||||
while self._running:
|
||||
current_url = self._stream_urls[url_idx % n]
|
||||
self._active_url = current_url
|
||||
try:
|
||||
await self._pump_once()
|
||||
await self._pump_once(current_url)
|
||||
url_idx = 0 # reset to primary after successful pump
|
||||
break
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
@@ -269,18 +295,31 @@ class BroadcastGroup:
|
||||
break
|
||||
consecutive += 1
|
||||
self._reconnect_count += 1
|
||||
wait = min(base_wait * (2 ** (consecutive - 1)), _BACKOFF_CAP)
|
||||
next_idx = (url_idx + 1) % n
|
||||
cycle_done = (consecutive % n == 0)
|
||||
|
||||
if n > 1:
|
||||
logger.warning(
|
||||
f"[ch{self.channel_id}] Domain {_url_netloc(current_url)} failed "
|
||||
f"(attempt {consecutive}), trying {_url_netloc(self._stream_urls[next_idx])}: {e}"
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
f"[ch{self.channel_id}] Upstream error (attempt {consecutive}): {e}"
|
||||
)
|
||||
|
||||
url_idx = next_idx
|
||||
self._status = "reconnecting"
|
||||
logger.warning(
|
||||
f"[ch{self.channel_id}] ffmpeg exited (attempt {consecutive}), "
|
||||
f"restarting in {wait:.1f}s: {e}"
|
||||
)
|
||||
# Drain queues so stale buffered chunks are not delivered
|
||||
# after the fresh ffmpeg process starts.
|
||||
async with self._lock:
|
||||
for handle in self._clients.values():
|
||||
_drain_queue(handle.queue)
|
||||
await asyncio.sleep(wait)
|
||||
|
||||
if cycle_done:
|
||||
# All domains tried — drain queues and back off before next cycle
|
||||
wait = min(base_wait * (2 ** (consecutive // n - 1)), _BACKOFF_CAP)
|
||||
logger.warning(f"[ch{self.channel_id}] All {n} domain(s) failed, retrying in {wait:.1f}s")
|
||||
async with self._lock:
|
||||
for handle in self._clients.values():
|
||||
_drain_queue(handle.queue)
|
||||
await asyncio.sleep(wait)
|
||||
# else: try next domain immediately (no sleep)
|
||||
finally:
|
||||
self._status = "stopped"
|
||||
self._running = False
|
||||
@@ -292,7 +331,7 @@ class BroadcastGroup:
|
||||
except asyncio.QueueFull:
|
||||
pass
|
||||
|
||||
async def _pump_once(self) -> None:
|
||||
async def _pump_once(self, url: str | None = None) -> None:
|
||||
"""
|
||||
Spawn ffmpeg in copy (remux) mode and pump its stdout to all client
|
||||
queues until ffmpeg exits or self._running goes False.
|
||||
@@ -302,6 +341,7 @@ class BroadcastGroup:
|
||||
to clients. PTS/DTS values are normalised by ffmpeg across reconnects,
|
||||
eliminating A/V desync entirely.
|
||||
"""
|
||||
stream_url = url if url is not None else self.stream_url
|
||||
cmd = [
|
||||
_FFMPEG,
|
||||
"-hide_banner", "-loglevel", "warning",
|
||||
@@ -316,7 +356,7 @@ class BroadcastGroup:
|
||||
# stops sending bytes for STALL_TIMEOUT seconds ffmpeg closes
|
||||
# the connection and tries to reconnect (or exits if it gives up).
|
||||
"-timeout", str(int(STALL_TIMEOUT * 1_000_000)),
|
||||
"-i", self.stream_url,
|
||||
"-i", stream_url,
|
||||
"-c", "copy", # remux only — zero transcoding, bit-for-bit quality
|
||||
"-f", "mpegts",
|
||||
"pipe:1",
|
||||
|
||||
Reference in New Issue
Block a user