feat: domain health check automático con stickiness de dominio

- health_checker: usa TCP connect en lugar de Xtream API para dominios CDN
  (correcto para dominios que solo sirven streams, no la API de credenciales)
  Intervalo reducido a 3 min; detecta recuperación automática y lo registra en log
- pool: _build_stream_urls excluye dominios con status='error'; si todos están caídos
  usa todos como fallback para no dejar el stream sin opciones
- restream: stickiness de dominio — en renovación de sesión normal reconecta al mismo
  dominio siempre; solo rota al siguiente en fallos reales (<25s)
- restream: _mark_domain_down marca el dominio como error en BD inmediatamente al
  detectar un fallo rápido, sin esperar al próximo ciclo del health checker
- database: usa ADD COLUMN IF NOT EXISTS para la migración (evita abortar transacción)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
KiraStream
2026-05-19 15:42:37 +00:00
parent 4bfc7592a6
commit 99155271bb
4 changed files with 189 additions and 69 deletions
+75 -50
View File
@@ -426,13 +426,43 @@ class BroadcastGroup:
except Exception as e:
logger.debug(f"probe_info failed for ch{self.channel_id}: {e}")
async def _mark_domain_down(self, url: str) -> None:
"""Update ProviderUrl.status='error' immediately so other streams skip this domain."""
from urllib.parse import urlparse
from datetime import datetime, timezone
from ..database import AsyncSessionLocal
from ..models.provider import ProviderUrl
from sqlalchemy import select
netloc = urlparse(url).netloc
if not netloc:
return
try:
async with AsyncSessionLocal() as db:
result = await db.execute(
select(ProviderUrl).where(
ProviderUrl.provider_account_id == self.provider_account_id,
ProviderUrl.url.ilike(f"%{netloc}%"),
)
)
pu = result.scalar_one_or_none()
if pu and pu.status != "error":
pu.status = "error"
pu.last_checked_at = datetime.now(timezone.utc)
await db.commit()
logger.info(f"[ch{self.channel_id}] Dominio {netloc} marcado como error (fallo rápido)")
except Exception as exc:
logger.debug(f"[ch{self.channel_id}] _mark_domain_down failed: {exc}")
async def _pump_with_retry(self) -> None:
# Failure-mode tracking: a "real failure" is any pump that exits in
# less than PROVIDER_SESSION_MIN seconds. Longer pumps are normal
# provider session timeouts (~2 min) and are invisible to the user.
in_failure_mode = False # True once at least one real failure has occurred
# Domain stickiness: on normal provider session timeouts (>PROVIDER_SESSION_MIN)
# we reconnect to the same domain — providers rotate DNS/CDN themselves, so
# staying on the same host avoids unnecessary domain cycling.
# Only on real quick failures (<PROVIDER_SESSION_MIN) do we advance to the
# next domain and immediately mark the failed one as 'error' in the DB.
in_failure_mode = False
failure_start: float | None = None
failure_attempts = 0 # consecutive real failures in this failure episode
failure_attempts = 0
preferred_idx = 0 # index of last domain that served a full session
base_wait = 0.5
url_idx = 0
n = len(self._stream_urls) or 1
@@ -444,7 +474,7 @@ class BroadcastGroup:
pump_start = time.time()
try:
await self._pump_once(current_url)
url_idx = 0 # reset to primary after successful pump
url_idx = 0
break
except asyncio.CancelledError:
raise
@@ -457,75 +487,70 @@ class BroadcastGroup:
is_real_failure = pump_duration < PROVIDER_SESSION_MIN
if is_real_failure:
# Quick exit → real provider/network error
# Quick exit → real error (DNS, refused, HTTP 4xx, stall)
self._failure_count += 1
failure_attempts += 1
cause = _classify_error(e, pump_duration)
if not in_failure_mode:
# First real failure of this episode → log caída
in_failure_mode = True
failure_start = time.time()
asyncio.create_task(self._log_stream_event(
"caida",
_url_netloc(current_url),
cause,
1,
int(pump_duration),
"caida", _url_netloc(current_url), cause, 1, int(pump_duration),
))
logger.warning(
f"[ch{self.channel_id}] CAÍDA — {cause} "
f"(pump={pump_duration:.1f}s)"
)
logger.warning(f"[ch{self.channel_id}] CAÍDA — {cause} (pump={pump_duration:.1f}s)")
else:
# Still failing — don't spam the log table
logger.warning(
f"[ch{self.channel_id}] Fallo continuado intento {failure_attempts} — "
f"{cause} (pump={pump_duration:.1f}s)"
)
else:
# Long pump → normal provider session timeout
self._session_count += 1
if in_failure_mode:
# Was in failure mode, now ran long → stream recovered
outage_secs = int(time.time() - (failure_start or time.time()))
asyncio.create_task(self._log_stream_event(
"recuperado",
_url_netloc(current_url),
f"Restablecido tras {failure_attempts} intento(s) y {outage_secs}s de interrupción",
failure_attempts,
outage_secs,
))
logger.info(
f"[ch{self.channel_id}] RECUPERADO — tras {failure_attempts} "
f"intento(s) en {outage_secs}s"
)
in_failure_mode = False
failure_start = None
failure_attempts = 0
else:
# Healthy session timeout — normal, no event logged
logger.info(
f"[ch{self.channel_id}] Sesión del proveedor cerrada tras "
f"{pump_duration:.1f}s (normal, sesión #{self._session_count})"
)
# Mark domain as error immediately so other streams skip it
asyncio.create_task(self._mark_domain_down(current_url))
next_idx = (url_idx + 1) % n
url_idx = next_idx
self._status = "reconnecting"
# Rotate to next domain
url_idx = (url_idx + 1) % n
if is_real_failure:
cycle_done = (failure_attempts % n == 0)
self._status = "reconnecting"
if cycle_done:
# All domains tried — back off before next cycle
wait = min(base_wait * (2 ** (failure_attempts // n - 1)), _BACKOFF_CAP)
logger.warning(
f"[ch{self.channel_id}] All {n} domain(s) failed "
f"(attempt {failure_attempts}), retrying in {wait:.1f}s"
)
await asyncio.sleep(wait)
# Normal session timeout: reconnect immediately (no sleep needed)
# else: try next domain immediately
else:
# Normal provider session timeout (>25 s) — reconnect to same domain
self._session_count += 1
preferred_idx = url_idx # remember: this domain works
url_idx = preferred_idx # stay on it
if in_failure_mode:
outage_secs = int(time.time() - (failure_start or time.time()))
asyncio.create_task(self._log_stream_event(
"recuperado", _url_netloc(current_url),
f"Restablecido tras {failure_attempts} intento(s) y {outage_secs}s de interrupción",
failure_attempts, outage_secs,
))
logger.info(
f"[ch{self.channel_id}] RECUPERADO — tras {failure_attempts} "
f"intento(s) en {outage_secs}s, dominio estable: {_url_netloc(current_url)}"
)
in_failure_mode = False
failure_start = None
failure_attempts = 0
else:
logger.info(
f"[ch{self.channel_id}] Sesión del proveedor cerrada tras "
f"{pump_duration:.1f}s (normal, sesión #{self._session_count}, "
f"dominio: {_url_netloc(current_url)})"
)
self._status = "reconnecting"
# Reconnect immediately — no backoff for normal session rotations
finally:
self._status = "stopped"