Files
KiraTV/backend/core/health_checker.py
T
joaquin cccdf9f137 feat: multi-domain failover for provider accounts
Each ProviderAccount can now have multiple base URLs (provider_urls table).
On stream failure, BroadcastGroup cycles to the next domain immediately
with no wait; backs off only after all domains have been tried once.
Background health checker pings every domain every 5 min via player_api.php
and updates status/response_ms. Admin UI shows domain list with color-coded
status badges and a "Verificar todos" button per provider.
2026-05-18 00:11:06 +02:00

116 lines
4.3 KiB
Python

"""Background health checker for provider URLs."""
import asyncio
import logging
import time
from datetime import datetime, timezone
import aiohttp
from sqlalchemy import select
from ..database import AsyncSessionLocal
from ..models.provider import ProviderAccount, ProviderUrl
logger = logging.getLogger(__name__)
CHECK_INTERVAL = 300 # 5 minutes
async def _check_url(base_url: str, username: str, password: str) -> tuple[str, int | None]:
url = f"{base_url.rstrip('/')}/player_api.php"
params = {"username": username, "password": password, "action": "user_info"}
start = time.time()
try:
timeout = aiohttp.ClientTimeout(total=8)
async with aiohttp.ClientSession(timeout=timeout) as session:
async with session.get(url, params=params, ssl=False) as resp:
elapsed_ms = int((time.time() - start) * 1000)
if resp.status == 200:
try:
data = await resp.json(content_type=None)
if "user_info" in data:
return "ok", elapsed_ms
except Exception:
pass
return "error", elapsed_ms
except asyncio.TimeoutError:
return "timeout", None
except Exception:
return "error", None
async def check_provider_urls(provider_account_id: int) -> list[dict]:
"""Check all URLs for a specific provider. Returns list of result dicts."""
async with AsyncSessionLocal() as db:
result = await db.execute(
select(ProviderUrl.id, ProviderUrl.url, ProviderAccount.username, ProviderAccount.password)
.join(ProviderAccount)
.where(
ProviderUrl.provider_account_id == provider_account_id,
ProviderUrl.is_active == True, # noqa: E712
)
)
rows = result.all()
if not rows:
return []
tasks = [_check_url(url, username, password) for _, url, username, password in rows]
results = await asyncio.gather(*tasks, return_exceptions=True)
now = datetime.now(timezone.utc)
output = []
async with AsyncSessionLocal() as db:
for (url_id, url, _, _), check_result in zip(rows, results):
if isinstance(check_result, Exception):
status, response_ms = "error", None
else:
status, response_ms = check_result
pu = await db.get(ProviderUrl, url_id)
if pu:
pu.status = status
pu.response_ms = response_ms
pu.last_checked_at = now
output.append({"id": url_id, "url": url, "status": status, "response_ms": response_ms})
await db.commit()
return output
async def health_check_loop() -> None:
"""Runs forever, checking all active provider URLs every CHECK_INTERVAL seconds."""
logger.info("Provider URL health checker started")
while True:
await asyncio.sleep(CHECK_INTERVAL)
try:
async with AsyncSessionLocal() as db:
result = await db.execute(
select(ProviderUrl.id, ProviderUrl.url, ProviderAccount.username, ProviderAccount.password)
.join(ProviderAccount)
.where(ProviderUrl.is_active == True) # noqa: E712
)
rows = result.all()
if not rows:
continue
tasks = [_check_url(url, username, password) for _, url, username, password in rows]
check_results = await asyncio.gather(*tasks, return_exceptions=True)
now = datetime.now(timezone.utc)
async with AsyncSessionLocal() as db:
for (url_id, _, _, _), check_result in zip(rows, check_results):
if isinstance(check_result, Exception):
status, response_ms = "error", None
else:
status, response_ms = check_result
pu = await db.get(ProviderUrl, url_id)
if pu:
pu.status = status
pu.response_ms = response_ms
pu.last_checked_at = now
await db.commit()
logger.info(f"Health check done: {len(rows)} URLs checked")
except Exception as e:
logger.error(f"Health check loop error: {e}")