fix: distinguir sesiones normales del proveedor de fallos reales en logs y dashboard

- Solo registra "caída" cuando ffmpeg sale en <25s (error real: DNS, 403, auth, stall)
- Timeouts de sesión del proveedor (~2 min) NO se registran como fallos — son normales
- "Recuperado" se registra cuando el stream vuelve a funcionar tras un fallo real
- Añade clasificación automática de causa: conexión rechazada, DNS, 403, timeout...
- Dashboard: "reinicios" ahora distingue sesiones del proveedor (gris, normal) vs fallos (rojo)
- Registro: nueva columna "Duración fallo", proveedor+dominio juntos, filas con color de fondo
- StreamEvent: campo pump_duration_secs para persistir duración del fallo o interrupción

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
KiraStream
2026-05-19 15:12:04 +00:00
parent f965ccafa8
commit 4bfc7592a6
12 changed files with 474 additions and 362 deletions
+1
View File
@@ -174,6 +174,7 @@ def _serialize_event(ev: StreamEvent) -> dict:
"domain": ev.domain, "domain": ev.domain,
"error_message": ev.error_message, "error_message": ev.error_message,
"attempt_number": ev.attempt_number, "attempt_number": ev.attempt_number,
"pump_duration_secs": ev.pump_duration_secs,
"created_at": ev.created_at.isoformat() if ev.created_at else None, "created_at": ev.created_at.isoformat() if ev.created_at else None,
} }
+111 -45
View File
@@ -51,6 +51,12 @@ _SENTINEL = object() # signals end-of-stream to client queues
_FFMPEG = shutil.which("ffmpeg") or "ffmpeg" _FFMPEG = shutil.which("ffmpeg") or "ffmpeg"
# Pumps that run for less than this many seconds before exiting are classified
# as real failures (connection refused, DNS error, auth expired, etc.).
# Pumps that run longer are normal provider session timeouts (~2 min) and are
# NOT logged as failures — they are transparent to the user thanks to the buffer.
PROVIDER_SESSION_MIN = 25.0
# ffmpeg stderr patterns that indicate input-side A/V issues (demuxer warnings). # ffmpeg stderr patterns that indicate input-side A/V issues (demuxer warnings).
# These don't affect output quality with -use_wallclock_as_timestamps but are # These don't affect output quality with -use_wallclock_as_timestamps but are
# counted as a health metric to surface in the dashboard. # counted as a health metric to surface in the dashboard.
@@ -66,6 +72,28 @@ _HEALTH_WARN_SECS = 12.0 # > 12 s without data → warning (normal reconnects
_HEALTH_ERROR_SECS = 25.0 # > 25 s → error, trigger correction _HEALTH_ERROR_SECS = 25.0 # > 25 s → error, trigger correction
def _classify_error(exc: Exception, pump_duration: float) -> str:
"""Return a human-readable Spanish cause for a stream failure."""
msg = str(exc).lower()
if "connection refused" in msg or "errno 111" in msg:
return "Conexión rechazada por el servidor del proveedor"
if "name or service not known" in msg or "could not resolve" in msg or "getaddrinfo" in msg:
return "No se pudo resolver el dominio (DNS)"
if "no output" in msg or ("timeout" in msg and "no" in msg):
return f"Sin datos del proveedor durante {int(pump_duration)}s (stall)"
if "403" in msg:
return "Acceso denegado por el proveedor (403 Forbidden)"
if "404" in msg:
return "Canal no encontrado en el proveedor (404)"
if "401" in msg:
return "Autenticación rechazada por el proveedor (401)"
if "rc=1" in msg or "rc=-1" in msg:
return f"Error del proveedor (ffmpeg rc≠0) tras {int(pump_duration)}s"
if "rc=0" in msg:
return f"Sesión del proveedor cerrada tras {int(pump_duration)}s"
return f"Error en {int(pump_duration)}s: {str(exc)[:200]}"
def _url_netloc(url: str) -> str: def _url_netloc(url: str) -> str:
from urllib.parse import urlparse from urllib.parse import urlparse
try: try:
@@ -134,6 +162,8 @@ class BroadcastGroup:
# Health / metadata # Health / metadata
self.stream_info: dict = {} self.stream_info: dict = {}
self._reconnect_count: int = 0 self._reconnect_count: int = 0
self._session_count: int = 0 # normal provider session timeouts (>PROVIDER_SESSION_MIN)
self._failure_count: int = 0 # real quick failures (<PROVIDER_SESSION_MIN)
self._status: str = "connecting" self._status: str = "connecting"
self._last_data_at: float = time.time() self._last_data_at: float = time.time()
@@ -251,6 +281,8 @@ class BroadcastGroup:
"running": self._running, "running": self._running,
"status": self._status, "status": self._status,
"reconnect_count": self._reconnect_count, "reconnect_count": self._reconnect_count,
"session_count": self._session_count,
"failure_count": self._failure_count,
"last_data_ago": round(time.time() - self._last_data_at, 1), "last_data_ago": round(time.time() - self._last_data_at, 1),
"stream_info": self.stream_info, "stream_info": self.stream_info,
# ffmpeg process info # ffmpeg process info
@@ -364,6 +396,7 @@ class BroadcastGroup:
domain: str, domain: str,
error_message: str, error_message: str,
attempt: int, attempt: int,
pump_duration_secs: int | None = None,
) -> None: ) -> None:
"""Persist a stream failure/recovery event to the database (fire-and-forget).""" """Persist a stream failure/recovery event to the database (fire-and-forget)."""
try: try:
@@ -378,6 +411,7 @@ class BroadcastGroup:
domain=domain, domain=domain,
error_message=error_message[:500] if error_message else None, error_message=error_message[:500] if error_message else None,
attempt_number=attempt, attempt_number=attempt,
pump_duration_secs=pump_duration_secs,
) )
db.add(ev) db.add(ev)
await db.commit() await db.commit()
@@ -393,10 +427,15 @@ class BroadcastGroup:
logger.debug(f"probe_info failed for ch{self.channel_id}: {e}") logger.debug(f"probe_info failed for ch{self.channel_id}: {e}")
async def _pump_with_retry(self) -> None: async def _pump_with_retry(self) -> None:
consecutive = 0 # Failure-mode tracking: a "real failure" is any pump that exits in
base_wait = 0.5 # less than PROVIDER_SESSION_MIN seconds. Longer pumps are normal
url_idx = 0 # provider session timeouts (~2 min) and are invisible to the user.
n = len(self._stream_urls) or 1 in_failure_mode = False # True once at least one real failure has occurred
failure_start: float | None = None
failure_attempts = 0 # consecutive real failures in this failure episode
base_wait = 0.5
url_idx = 0
n = len(self._stream_urls) or 1
try: try:
while self._running: while self._running:
@@ -413,54 +452,81 @@ class BroadcastGroup:
if not self._running: if not self._running:
break break
pump_duration = time.time() - pump_start pump_duration = time.time() - pump_start
# Recovery detection: had failures before but this attempt ran stably
# for >30s — the stream was healthy and dropped again (e.g. provider
# session timeout). Log recovery, then treat next failure as fresh drop.
if consecutive > 1 and pump_duration > 30:
asyncio.create_task(self._log_stream_event(
"recuperado",
_url_netloc(current_url),
f"Estable {int(pump_duration)}s antes de caer de nuevo",
consecutive,
))
consecutive = 0 # reset so next failure is logged as a fresh "caida"
consecutive += 1
self._reconnect_count += 1 self._reconnect_count += 1
next_idx = (url_idx + 1) % n
cycle_done = (consecutive % n == 0)
# Log the first drop of each failure sequence is_real_failure = pump_duration < PROVIDER_SESSION_MIN
if consecutive == 1:
asyncio.create_task(self._log_stream_event(
"caida",
_url_netloc(current_url),
str(e),
1,
))
if n > 1: if is_real_failure:
logger.warning( # Quick exit → real provider/network error
f"[ch{self.channel_id}] Domain {_url_netloc(current_url)} failed " self._failure_count += 1
f"(attempt {consecutive}), trying {_url_netloc(self._stream_urls[next_idx])}: {e}" failure_attempts += 1
) cause = _classify_error(e, pump_duration)
if not in_failure_mode:
# First real failure of this episode → log caída
in_failure_mode = True
failure_start = time.time()
asyncio.create_task(self._log_stream_event(
"caida",
_url_netloc(current_url),
cause,
1,
int(pump_duration),
))
logger.warning(
f"[ch{self.channel_id}] CAÍDA — {cause} "
f"(pump={pump_duration:.1f}s)"
)
else:
# Still failing — don't spam the log table
logger.warning(
f"[ch{self.channel_id}] Fallo continuado intento {failure_attempts} — "
f"{cause} (pump={pump_duration:.1f}s)"
)
else: else:
logger.warning( # Long pump → normal provider session timeout
f"[ch{self.channel_id}] Upstream error (attempt {consecutive}): {e}" self._session_count += 1
)
url_idx = next_idx if in_failure_mode:
# Was in failure mode, now ran long → stream recovered
outage_secs = int(time.time() - (failure_start or time.time()))
asyncio.create_task(self._log_stream_event(
"recuperado",
_url_netloc(current_url),
f"Restablecido tras {failure_attempts} intento(s) y {outage_secs}s de interrupción",
failure_attempts,
outage_secs,
))
logger.info(
f"[ch{self.channel_id}] RECUPERADO — tras {failure_attempts} "
f"intento(s) en {outage_secs}s"
)
in_failure_mode = False
failure_start = None
failure_attempts = 0
else:
# Healthy session timeout — normal, no event logged
logger.info(
f"[ch{self.channel_id}] Sesión del proveedor cerrada tras "
f"{pump_duration:.1f}s (normal, sesión #{self._session_count})"
)
next_idx = (url_idx + 1) % n
url_idx = next_idx
self._status = "reconnecting" self._status = "reconnecting"
if cycle_done: if is_real_failure:
# All domains tried — back off before next cycle. cycle_done = (failure_attempts % n == 0)
# Do NOT drain client queues: the buffered data (up to 32 MB) covers if cycle_done:
# the reconnect window so players never see a black screen. # All domains tried — back off before next cycle
wait = min(base_wait * (2 ** (consecutive // n - 1)), _BACKOFF_CAP) wait = min(base_wait * (2 ** (failure_attempts // n - 1)), _BACKOFF_CAP)
logger.warning(f"[ch{self.channel_id}] All {n} domain(s) failed, retrying in {wait:.1f}s") logger.warning(
await asyncio.sleep(wait) f"[ch{self.channel_id}] All {n} domain(s) failed "
# else: try next domain immediately (no sleep) f"(attempt {failure_attempts}), retrying in {wait:.1f}s"
)
await asyncio.sleep(wait)
# Normal session timeout: reconnect immediately (no sleep needed)
finally: finally:
self._status = "stopped" self._status = "stopped"
self._running = False self._running = False
+7
View File
@@ -31,6 +31,13 @@ async def init_db():
from .models.provider import ProviderUrl # noqa: ensure ProviderUrl is registered from .models.provider import ProviderUrl # noqa: ensure ProviderUrl is registered
async with engine.begin() as conn: async with engine.begin() as conn:
await conn.run_sync(Base.metadata.create_all) await conn.run_sync(Base.metadata.create_all)
# Migrate: add pump_duration_secs column if it doesn't exist yet
try:
await conn.execute(text(
"ALTER TABLE stream_events ADD COLUMN pump_duration_secs INTEGER"
))
except Exception:
pass # column already exists
# Migrate: create provider_urls entries from existing base_url if not already done # Migrate: create provider_urls entries from existing base_url if not already done
await conn.execute(text(""" await conn.execute(text("""
INSERT INTO provider_urls (provider_account_id, url, priority, is_active, status) INSERT INTO provider_urls (provider_account_id, url, priority, is_active, status)
+6 -5
View File
@@ -9,8 +9,9 @@ class StreamEvent(Base):
channel_id = Column(Integer, nullable=True, index=True) channel_id = Column(Integer, nullable=True, index=True)
channel_name = Column(String(500), nullable=True) channel_name = Column(String(500), nullable=True)
provider_name = Column(String(200), nullable=True) provider_name = Column(String(200), nullable=True)
event_type = Column(String(50), nullable=False, index=True) # caida | recuperado event_type = Column(String(50), nullable=False, index=True) # caida | recuperado
domain = Column(String(300), nullable=True) domain = Column(String(300), nullable=True)
error_message = Column(Text, nullable=True) error_message = Column(Text, nullable=True)
attempt_number = Column(Integer, nullable=True) attempt_number = Column(Integer, nullable=True)
created_at = Column(DateTime(timezone=True), server_default=func.now(), index=True) pump_duration_secs = Column(Integer, nullable=True) # seconds ffmpeg ran (caída) or outage length (recuperado)
created_at = Column(DateTime(timezone=True), server_default=func.now(), index=True)
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.3 MiB

+2 -2
View File
@@ -5,8 +5,8 @@
<meta name="viewport" content="width=device-width, initial-scale=1.0" /> <meta name="viewport" content="width=device-width, initial-scale=1.0" />
<title>KiraStream Admin</title> <title>KiraStream Admin</title>
<link rel="icon" type="image/svg+xml" href="/favicon.svg" /> <link rel="icon" type="image/svg+xml" href="/favicon.svg" />
<script type="module" crossorigin src="/panel/assets/index-H1dlHjcY.js"></script> <script type="module" crossorigin src="/panel/assets/index-BMreIS4S.js"></script>
<link rel="stylesheet" crossorigin href="/panel/assets/index-Bf2TNXpE.css"> <link rel="stylesheet" crossorigin href="/panel/assets/index-W-3W5628.css">
</head> </head>
<body> <body>
<div id="root"></div> <div id="root"></div>
+42 -15
View File
@@ -34,7 +34,9 @@ interface StreamSession {
running: boolean running: boolean
// stream health // stream health
status?: 'connecting' | 'ok' | 'reconnecting' | 'stopped' status?: 'connecting' | 'ok' | 'reconnecting' | 'stopped'
reconnect_count?: number reconnect_count?: number // total ffmpeg restarts (sessions + failures)
session_count?: number // normal provider session timeouts (expected, no interruption)
failure_count?: number // real quick failures (<25s) — actual errors
last_data_ago?: number last_data_ago?: number
stream_info?: StreamInfo stream_info?: StreamInfo
// ffmpeg process info // ffmpeg process info
@@ -119,25 +121,49 @@ function StreamInfoBadge({ info }: { info: StreamInfo }) {
) )
} }
function HealthBadge({ status, reconnectCount }: { function HealthBadge({ status, sessionCount, failureCount }: {
status?: string status?: string
reconnectCount?: number sessionCount?: number
failureCount?: number
}) { }) {
if (!status || status === 'ok') { const hasFail = failureCount != null && failureCount > 0
if (reconnectCount && reconnectCount > 0) { const hasSess = sessionCount != null && sessionCount > 0
return (
<span className="inline-flex items-center gap-1 text-[10px] px-1.5 py-0.5 rounded bg-yellow-900/40 text-yellow-300 border border-yellow-800" title={`${reconnectCount} reconexión(es)`}>
<RefreshCw size={9} /> {reconnectCount}
</span>
)
}
return <span className="inline-flex items-center gap-1 text-[10px] text-green-500"><span className="w-1.5 h-1.5 rounded-full bg-green-500 animate-pulse inline-block" /> OK</span>
}
if (status === 'connecting') { if (status === 'connecting') {
return <span className="inline-flex items-center gap-1 text-[10px] text-gray-400"><RefreshCw size={9} className="animate-spin" /> Conectando</span> return <span className="inline-flex items-center gap-1 text-[10px] text-gray-400"><RefreshCw size={9} className="animate-spin" /> Conectando</span>
} }
if (status === 'reconnecting') { if (status === 'reconnecting') {
return <span className="inline-flex items-center gap-1 text-[10px] text-orange-400"><RefreshCw size={9} className="animate-spin" /> Reconectando {reconnectCount ? `(×${reconnectCount})` : ''}</span> return (
<span className="inline-flex items-center gap-1 text-[10px] text-orange-400">
<RefreshCw size={9} className="animate-spin" /> Reconectando
{hasFail && <span className="ml-1 px-1 bg-red-900/40 text-red-300 border border-red-800 rounded">{failureCount} fallo{failureCount! > 1 ? 's' : ''}</span>}
</span>
)
}
if (!status || status === 'ok') {
return (
<div className="flex flex-col gap-0.5">
<span className="inline-flex items-center gap-1 text-[10px] text-green-500">
<span className="w-1.5 h-1.5 rounded-full bg-green-500 animate-pulse inline-block" /> OK
</span>
{hasSess && (
<span
className="inline-flex items-center gap-1 text-[10px] text-gray-500"
title={`${sessionCount} rotación(es) de sesión del proveedor — el proveedor cierra la conexión cada ~2 min (normal). El buffer absorbe el cambio sin interrupción.`}
>
<RefreshCw size={8} /> {sessionCount} sesión{sessionCount! > 1 ? 'es' : ''}
</span>
)}
{hasFail && (
<span
className="inline-flex items-center gap-1 text-[10px] px-1 bg-red-900/40 text-red-300 border border-red-800 rounded"
title={`${failureCount} fallo(s) real(es) detectado(s) — conexión rechazada, DNS, auth u otro error. Ver Registro → Fallos de stream.`}
>
<AlertTriangle size={8} /> {failureCount} fallo{failureCount! > 1 ? 's' : ''}
</span>
)}
</div>
)
} }
return <span className="inline-flex items-center gap-1 text-[10px] text-red-400"><AlertTriangle size={9} /> {status}</span> return <span className="inline-flex items-center gap-1 text-[10px] text-red-400"><AlertTriangle size={9} /> {status}</span>
} }
@@ -431,7 +457,8 @@ export default function Dashboard() {
<td className="px-4 py-2.5"> <td className="px-4 py-2.5">
<HealthBadge <HealthBadge
status={s.status} status={s.status}
reconnectCount={s.reconnect_count} sessionCount={s.session_count}
failureCount={s.failure_count}
/> />
{s.ffmpeg_pid && ( {s.ffmpeg_pid && (
<div className="mt-1.5 flex items-center gap-1.5"> <div className="mt-1.5 flex items-center gap-1.5">
+26 -16
View File
@@ -39,6 +39,7 @@ interface StreamEvent {
domain: string | null domain: string | null
error_message: string | null error_message: string | null
attempt_number: number | null attempt_number: number | null
pump_duration_secs: number | null // for caída: seconds ffmpeg ran; for recuperado: outage length
created_at: string | null created_at: string | null
} }
@@ -375,42 +376,51 @@ function EventsTab() {
<th className="text-left px-4 py-2">Fecha</th> <th className="text-left px-4 py-2">Fecha</th>
<th className="text-left px-4 py-2">Tipo</th> <th className="text-left px-4 py-2">Tipo</th>
<th className="text-left px-4 py-2">Canal</th> <th className="text-left px-4 py-2">Canal</th>
<th className="text-left px-4 py-2">Proveedor</th> <th className="text-left px-4 py-2">Proveedor · Dominio</th>
<th className="text-left px-4 py-2">Dominio</th> <th className="text-left px-4 py-2">Causa</th>
<th className="text-left px-4 py-2">Detalle</th> <th className="text-left px-4 py-2 whitespace-nowrap" title="Para caídas: tiempo que duró ffmpeg antes de fallar. Para recuperados: duración total de la interrupción.">Duración fallo</th>
<th className="text-right px-4 py-2">Intento</th> <th className="text-right px-4 py-2" title="Número de intentos fallidos antes de recuperar (solo en recuperados)">Intentos</th>
</tr> </tr>
</thead> </thead>
<tbody> <tbody>
{loading && <tr><td colSpan={7} className="text-center py-8 text-gray-500">Cargando...</td></tr>} {loading && <tr><td colSpan={7} className="text-center py-8 text-gray-500">Cargando...</td></tr>}
{!loading && events.length === 0 && ( {!loading && events.length === 0 && (
<tr><td colSpan={7} className="text-center py-8 text-gray-500">No hay eventos registrados — los fallos aparecerán aquí automáticamente</td></tr> <tr><td colSpan={7} className="text-center py-8 text-gray-500">Sin fallos registrados — los errores reales del proveedor aparecerán aquí automáticamente</td></tr>
)} )}
{events.map(ev => ( {events.map(ev => (
<tr key={ev.id} className="border-b border-dark-700/50 hover:bg-dark-700/30 transition-colors"> <tr key={ev.id} className={`border-b border-dark-700/50 hover:bg-dark-700/30 transition-colors ${ev.event_type === 'caida' ? 'bg-red-950/10' : ev.event_type === 'recuperado' ? 'bg-emerald-950/10' : ''}`}>
<td className="px-4 py-2 text-gray-400 text-xs whitespace-nowrap">{fmtDate(ev.created_at)}</td> <td className="px-4 py-2.5 text-gray-400 text-xs whitespace-nowrap">{fmtDate(ev.created_at)}</td>
<td className="px-4 py-2"> <td className="px-4 py-2.5">
{ev.event_type === 'caida' ? ( {ev.event_type === 'caida' ? (
<span className="inline-flex items-center gap-1 text-xs px-2 py-0.5 rounded-full bg-red-500/20 text-red-400"> <span className="inline-flex items-center gap-1 text-xs px-2 py-0.5 rounded-full bg-red-500/20 text-red-400 border border-red-800/50">
<AlertTriangle size={10} /> Caída <AlertTriangle size={10} /> Caída
</span> </span>
) : ev.event_type === 'recuperado' ? ( ) : ev.event_type === 'recuperado' ? (
<span className="inline-flex items-center gap-1 text-xs px-2 py-0.5 rounded-full bg-emerald-500/20 text-emerald-400"> <span className="inline-flex items-center gap-1 text-xs px-2 py-0.5 rounded-full bg-emerald-500/20 text-emerald-400 border border-emerald-800/50">
<CheckCircle size={10} /> Recuperado <CheckCircle size={10} /> Recuperado
</span> </span>
) : ( ) : (
<span className="text-xs text-gray-400">{ev.event_type}</span> <span className="text-xs text-gray-400">{ev.event_type}</span>
)} )}
</td> </td>
<td className="px-4 py-2 text-gray-200 max-w-[180px] truncate" title={ev.channel_name ?? ''}> <td className="px-4 py-2.5 text-gray-200 max-w-[180px] truncate text-xs" title={ev.channel_name ?? ''}>
{ev.channel_name || `#${ev.channel_id}` || '—'} {ev.channel_name || (ev.channel_id ? `#${ev.channel_id}` : '—')}
</td> </td>
<td className="px-4 py-2 text-gray-400 text-xs">{ev.provider_name || '—'}</td> <td className="px-4 py-2.5 text-xs">
<td className="px-4 py-2 text-gray-400 font-mono text-xs">{ev.domain || '—'}</td> <span className="text-gray-400">{ev.provider_name || '—'}</span>
<td className="px-4 py-2 text-gray-500 text-xs max-w-[240px] truncate" title={ev.error_message ?? ''}> {ev.domain && <span className="block text-gray-600 font-mono text-[10px]">{ev.domain}</span>}
</td>
<td className="px-4 py-2.5 text-gray-300 text-xs max-w-[260px]" title={ev.error_message ?? ''}>
{ev.error_message || '—'} {ev.error_message || '—'}
</td> </td>
<td className="px-4 py-2 text-right text-gray-500 text-xs tabular-nums"> <td className="px-4 py-2.5 text-xs tabular-nums">
{ev.pump_duration_secs != null ? (
<span className={ev.event_type === 'caida' ? 'text-red-400' : 'text-emerald-400'}>
{fmtDuration(ev.pump_duration_secs)}
</span>
) : '—'}
</td>
<td className="px-4 py-2.5 text-right text-gray-500 text-xs tabular-nums">
{ev.attempt_number ?? '—'} {ev.attempt_number ?? '—'}
</td> </td>
</tr> </tr>