feat: domain health check automático con stickiness de dominio
- health_checker: usa TCP connect en lugar de Xtream API para dominios CDN (correcto para dominios que solo sirven streams, no la API de credenciales) Intervalo reducido a 3 min; detecta recuperación automática y lo registra en log - pool: _build_stream_urls excluye dominios con status='error'; si todos están caídos usa todos como fallback para no dejar el stream sin opciones - restream: stickiness de dominio — en renovación de sesión normal reconecta al mismo dominio siempre; solo rota al siguiente en fallos reales (<25s) - restream: _mark_domain_down marca el dominio como error en BD inmediatamente al detectar un fallo rápido, sin esperar al próximo ciclo del health checker - database: usa ADD COLUMN IF NOT EXISTS para la migración (evita abortar transacción) Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
+75
-50
@@ -426,13 +426,43 @@ class BroadcastGroup:
|
||||
except Exception as e:
|
||||
logger.debug(f"probe_info failed for ch{self.channel_id}: {e}")
|
||||
|
||||
async def _mark_domain_down(self, url: str) -> None:
|
||||
"""Update ProviderUrl.status='error' immediately so other streams skip this domain."""
|
||||
from urllib.parse import urlparse
|
||||
from datetime import datetime, timezone
|
||||
from ..database import AsyncSessionLocal
|
||||
from ..models.provider import ProviderUrl
|
||||
from sqlalchemy import select
|
||||
netloc = urlparse(url).netloc
|
||||
if not netloc:
|
||||
return
|
||||
try:
|
||||
async with AsyncSessionLocal() as db:
|
||||
result = await db.execute(
|
||||
select(ProviderUrl).where(
|
||||
ProviderUrl.provider_account_id == self.provider_account_id,
|
||||
ProviderUrl.url.ilike(f"%{netloc}%"),
|
||||
)
|
||||
)
|
||||
pu = result.scalar_one_or_none()
|
||||
if pu and pu.status != "error":
|
||||
pu.status = "error"
|
||||
pu.last_checked_at = datetime.now(timezone.utc)
|
||||
await db.commit()
|
||||
logger.info(f"[ch{self.channel_id}] Dominio {netloc} marcado como error (fallo rápido)")
|
||||
except Exception as exc:
|
||||
logger.debug(f"[ch{self.channel_id}] _mark_domain_down failed: {exc}")
|
||||
|
||||
async def _pump_with_retry(self) -> None:
|
||||
# Failure-mode tracking: a "real failure" is any pump that exits in
|
||||
# less than PROVIDER_SESSION_MIN seconds. Longer pumps are normal
|
||||
# provider session timeouts (~2 min) and are invisible to the user.
|
||||
in_failure_mode = False # True once at least one real failure has occurred
|
||||
# Domain stickiness: on normal provider session timeouts (>PROVIDER_SESSION_MIN)
|
||||
# we reconnect to the same domain — providers rotate DNS/CDN themselves, so
|
||||
# staying on the same host avoids unnecessary domain cycling.
|
||||
# Only on real quick failures (<PROVIDER_SESSION_MIN) do we advance to the
|
||||
# next domain and immediately mark the failed one as 'error' in the DB.
|
||||
in_failure_mode = False
|
||||
failure_start: float | None = None
|
||||
failure_attempts = 0 # consecutive real failures in this failure episode
|
||||
failure_attempts = 0
|
||||
preferred_idx = 0 # index of last domain that served a full session
|
||||
base_wait = 0.5
|
||||
url_idx = 0
|
||||
n = len(self._stream_urls) or 1
|
||||
@@ -444,7 +474,7 @@ class BroadcastGroup:
|
||||
pump_start = time.time()
|
||||
try:
|
||||
await self._pump_once(current_url)
|
||||
url_idx = 0 # reset to primary after successful pump
|
||||
url_idx = 0
|
||||
break
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
@@ -457,75 +487,70 @@ class BroadcastGroup:
|
||||
is_real_failure = pump_duration < PROVIDER_SESSION_MIN
|
||||
|
||||
if is_real_failure:
|
||||
# Quick exit → real provider/network error
|
||||
# Quick exit → real error (DNS, refused, HTTP 4xx, stall)
|
||||
self._failure_count += 1
|
||||
failure_attempts += 1
|
||||
cause = _classify_error(e, pump_duration)
|
||||
|
||||
if not in_failure_mode:
|
||||
# First real failure of this episode → log caída
|
||||
in_failure_mode = True
|
||||
failure_start = time.time()
|
||||
asyncio.create_task(self._log_stream_event(
|
||||
"caida",
|
||||
_url_netloc(current_url),
|
||||
cause,
|
||||
1,
|
||||
int(pump_duration),
|
||||
"caida", _url_netloc(current_url), cause, 1, int(pump_duration),
|
||||
))
|
||||
logger.warning(
|
||||
f"[ch{self.channel_id}] CAÍDA — {cause} "
|
||||
f"(pump={pump_duration:.1f}s)"
|
||||
)
|
||||
logger.warning(f"[ch{self.channel_id}] CAÍDA — {cause} (pump={pump_duration:.1f}s)")
|
||||
else:
|
||||
# Still failing — don't spam the log table
|
||||
logger.warning(
|
||||
f"[ch{self.channel_id}] Fallo continuado intento {failure_attempts} — "
|
||||
f"{cause} (pump={pump_duration:.1f}s)"
|
||||
)
|
||||
else:
|
||||
# Long pump → normal provider session timeout
|
||||
self._session_count += 1
|
||||
|
||||
if in_failure_mode:
|
||||
# Was in failure mode, now ran long → stream recovered
|
||||
outage_secs = int(time.time() - (failure_start or time.time()))
|
||||
asyncio.create_task(self._log_stream_event(
|
||||
"recuperado",
|
||||
_url_netloc(current_url),
|
||||
f"Restablecido tras {failure_attempts} intento(s) y {outage_secs}s de interrupción",
|
||||
failure_attempts,
|
||||
outage_secs,
|
||||
))
|
||||
logger.info(
|
||||
f"[ch{self.channel_id}] RECUPERADO — tras {failure_attempts} "
|
||||
f"intento(s) en {outage_secs}s"
|
||||
)
|
||||
in_failure_mode = False
|
||||
failure_start = None
|
||||
failure_attempts = 0
|
||||
else:
|
||||
# Healthy session timeout — normal, no event logged
|
||||
logger.info(
|
||||
f"[ch{self.channel_id}] Sesión del proveedor cerrada tras "
|
||||
f"{pump_duration:.1f}s (normal, sesión #{self._session_count})"
|
||||
)
|
||||
# Mark domain as error immediately so other streams skip it
|
||||
asyncio.create_task(self._mark_domain_down(current_url))
|
||||
|
||||
next_idx = (url_idx + 1) % n
|
||||
url_idx = next_idx
|
||||
self._status = "reconnecting"
|
||||
# Rotate to next domain
|
||||
url_idx = (url_idx + 1) % n
|
||||
|
||||
if is_real_failure:
|
||||
cycle_done = (failure_attempts % n == 0)
|
||||
self._status = "reconnecting"
|
||||
if cycle_done:
|
||||
# All domains tried — back off before next cycle
|
||||
wait = min(base_wait * (2 ** (failure_attempts // n - 1)), _BACKOFF_CAP)
|
||||
logger.warning(
|
||||
f"[ch{self.channel_id}] All {n} domain(s) failed "
|
||||
f"(attempt {failure_attempts}), retrying in {wait:.1f}s"
|
||||
)
|
||||
await asyncio.sleep(wait)
|
||||
# Normal session timeout: reconnect immediately (no sleep needed)
|
||||
# else: try next domain immediately
|
||||
|
||||
else:
|
||||
# Normal provider session timeout (>25 s) — reconnect to same domain
|
||||
self._session_count += 1
|
||||
preferred_idx = url_idx # remember: this domain works
|
||||
url_idx = preferred_idx # stay on it
|
||||
|
||||
if in_failure_mode:
|
||||
outage_secs = int(time.time() - (failure_start or time.time()))
|
||||
asyncio.create_task(self._log_stream_event(
|
||||
"recuperado", _url_netloc(current_url),
|
||||
f"Restablecido tras {failure_attempts} intento(s) y {outage_secs}s de interrupción",
|
||||
failure_attempts, outage_secs,
|
||||
))
|
||||
logger.info(
|
||||
f"[ch{self.channel_id}] RECUPERADO — tras {failure_attempts} "
|
||||
f"intento(s) en {outage_secs}s, dominio estable: {_url_netloc(current_url)}"
|
||||
)
|
||||
in_failure_mode = False
|
||||
failure_start = None
|
||||
failure_attempts = 0
|
||||
else:
|
||||
logger.info(
|
||||
f"[ch{self.channel_id}] Sesión del proveedor cerrada tras "
|
||||
f"{pump_duration:.1f}s (normal, sesión #{self._session_count}, "
|
||||
f"dominio: {_url_netloc(current_url)})"
|
||||
)
|
||||
|
||||
self._status = "reconnecting"
|
||||
# Reconnect immediately — no backoff for normal session rotations
|
||||
|
||||
finally:
|
||||
self._status = "stopped"
|
||||
|
||||
Reference in New Issue
Block a user