feat: multi-domain failover for provider accounts

Each ProviderAccount can now have multiple base URLs (provider_urls table).
On stream failure, BroadcastGroup cycles to the next domain immediately
with no wait; backs off only after all domains have been tried once.
Background health checker pings every domain every 5 min via player_api.php
and updates status/response_ms. Admin UI shows domain list with color-coded
status badges and a "Verificar todos" button per provider.
This commit is contained in:
joaquin
2026-05-18 00:11:06 +02:00
parent 3a450214ab
commit cccdf9f137
8 changed files with 567 additions and 86 deletions
+56 -16
View File
@@ -51,6 +51,14 @@ _SENTINEL = object() # signals end-of-stream to client queues
_FFMPEG = shutil.which("ffmpeg") or "ffmpeg"
def _url_netloc(url: str) -> str:
from urllib.parse import urlparse
try:
return urlparse(url).netloc
except Exception:
return url
@dataclass
class ClientHandle:
client_id: str
@@ -75,14 +83,24 @@ class BroadcastGroup:
def __init__(
self,
channel_id: int,
stream_url: str,
stream_urls: list[str],
provider_account_id: int,
channel_name: str = "",
provider_name: str = "",
# kept for backward compat in callers that still pass stream_url as kwarg
stream_url: str | None = None,
):
self.channel_id = channel_id
self.channel_name = channel_name
self.stream_url = stream_url
# Support old callers that pass a single stream_url
if stream_urls:
self._stream_urls = stream_urls
elif stream_url:
self._stream_urls = [stream_url]
else:
self._stream_urls = []
self.stream_url = self._stream_urls[0] if self._stream_urls else ""
self._active_url: str = self.stream_url
self.provider_account_id = provider_account_id
self.provider_name = provider_name
self.started_at = time.time()
@@ -212,6 +230,9 @@ class BroadcastGroup:
# ffmpeg process info
"ffmpeg_pid": self._ffmpeg_pid,
"ffmpeg_cpu_pct": round(self._ffmpeg_cpu, 2),
# multi-domain failover info
"active_url_domain": _url_netloc(self._active_url),
"url_count": len(self._stream_urls),
}
# ------------------------------------------------------------------
@@ -256,11 +277,16 @@ class BroadcastGroup:
async def _pump_with_retry(self) -> None:
consecutive = 0
base_wait = 0.5
url_idx = 0
n = len(self._stream_urls) or 1
try:
while self._running:
current_url = self._stream_urls[url_idx % n]
self._active_url = current_url
try:
await self._pump_once()
await self._pump_once(current_url)
url_idx = 0 # reset to primary after successful pump
break
except asyncio.CancelledError:
raise
@@ -269,18 +295,31 @@ class BroadcastGroup:
break
consecutive += 1
self._reconnect_count += 1
wait = min(base_wait * (2 ** (consecutive - 1)), _BACKOFF_CAP)
next_idx = (url_idx + 1) % n
cycle_done = (consecutive % n == 0)
if n > 1:
logger.warning(
f"[ch{self.channel_id}] Domain {_url_netloc(current_url)} failed "
f"(attempt {consecutive}), trying {_url_netloc(self._stream_urls[next_idx])}: {e}"
)
else:
logger.warning(
f"[ch{self.channel_id}] Upstream error (attempt {consecutive}): {e}"
)
url_idx = next_idx
self._status = "reconnecting"
logger.warning(
f"[ch{self.channel_id}] ffmpeg exited (attempt {consecutive}), "
f"restarting in {wait:.1f}s: {e}"
)
# Drain queues so stale buffered chunks are not delivered
# after the fresh ffmpeg process starts.
async with self._lock:
for handle in self._clients.values():
_drain_queue(handle.queue)
await asyncio.sleep(wait)
if cycle_done:
# All domains tried — drain queues and back off before next cycle
wait = min(base_wait * (2 ** (consecutive // n - 1)), _BACKOFF_CAP)
logger.warning(f"[ch{self.channel_id}] All {n} domain(s) failed, retrying in {wait:.1f}s")
async with self._lock:
for handle in self._clients.values():
_drain_queue(handle.queue)
await asyncio.sleep(wait)
# else: try next domain immediately (no sleep)
finally:
self._status = "stopped"
self._running = False
@@ -292,7 +331,7 @@ class BroadcastGroup:
except asyncio.QueueFull:
pass
async def _pump_once(self) -> None:
async def _pump_once(self, url: str | None = None) -> None:
"""
Spawn ffmpeg in copy (remux) mode and pump its stdout to all client
queues until ffmpeg exits or self._running goes False.
@@ -302,6 +341,7 @@ class BroadcastGroup:
to clients. PTS/DTS values are normalised by ffmpeg across reconnects,
eliminating A/V desync entirely.
"""
stream_url = url if url is not None else self.stream_url
cmd = [
_FFMPEG,
"-hide_banner", "-loglevel", "warning",
@@ -316,7 +356,7 @@ class BroadcastGroup:
# stops sending bytes for STALL_TIMEOUT seconds ffmpeg closes
# the connection and tries to reconnect (or exits if it gives up).
"-timeout", str(int(STALL_TIMEOUT * 1_000_000)),
"-i", self.stream_url,
"-i", stream_url,
"-c", "copy", # remux only — zero transcoding, bit-for-bit quality
"-f", "mpegts",
"pipe:1",