Refactored server/db.js and all routes to use asynchronous fs operations, refactored m3uParser and epgParser to use streaming (via readline and sax) to prevent errors with large files, updated server/routes/proxy.js to use native stream piping for images and video, reducing latency and memory overhead

This commit is contained in:
Trevor Mears
2025-12-30 00:34:55 -08:00
parent 3b0d311756
commit d63c9c37d7
9 changed files with 366 additions and 311 deletions
+137 -104
View File
@@ -1,14 +1,11 @@
/**
* EPG (XMLTV) Parser
* Parses XMLTV format EPG data and extracts channel/programme information
* EPG (XMLTV) Parser (Streaming)
* Parses XMLTV format EPG data and extracts channel/programme information using streaming XML parser
*/
const { parseString } = require('xml2js');
const { promisify } = require('util');
const sax = require('sax');
const zlib = require('zlib');
const parseXml = promisify(parseString);
const gunzip = promisify(zlib.gunzip);
const { Readable } = require('stream');
/**
* Parse XMLTV date format (YYYYMMDDHHmmss +ZZZZ)
@@ -40,96 +37,120 @@ function parseXmltvDate(dateStr) {
}
/**
* Parse XMLTV content
* @param {string} content - Raw XMLTV content
* Parse XMLTV content (Stream or String)
* @param {Readable|string} input - XMLTV content as Stream or String
* @returns {Promise<{ channels: Array, programmes: Array }>}
*/
async function parse(content) {
const result = await parseXml(content, {
explicitArray: false,
mergeAttrs: true
});
function parse(input) {
return new Promise((resolve, reject) => {
const channels = [];
const programmes = [];
if (!result.tv) {
throw new Error('Invalid XMLTV format: missing <tv> root element');
}
const saxStream = sax.createStream(true, { trim: true, normalize: true }); // strict mode
const tv = result.tv;
const channels = [];
const programmes = [];
let currentTag = null;
let currentObject = null;
let textBuffer = '';
// Parse channels
const channelList = Array.isArray(tv.channel) ? tv.channel : (tv.channel ? [tv.channel] : []);
for (const ch of channelList) {
const channel = {
id: ch.id,
name: extractText(ch['display-name']),
icon: ch.icon ? (ch.icon.src || ch.icon) : null,
url: extractText(ch.url)
};
channels.push(channel);
}
saxStream.on('error', function (e) {
// clear the error
this._parser.error = null;
this._parser.resume();
console.warn('XML Parse Warning:', e.message);
});
// Parse programmes
const programmeList = Array.isArray(tv.programme) ? tv.programme : (tv.programme ? [tv.programme] : []);
for (const prog of programmeList) {
const programme = {
channelId: prog.channel,
start: parseXmltvDate(prog.start),
stop: parseXmltvDate(prog.stop),
title: extractText(prog.title),
subtitle: extractText(prog['sub-title']),
description: extractText(prog.desc),
category: extractCategories(prog.category),
icon: prog.icon ? (prog.icon.src || prog.icon) : null,
date: extractText(prog.date),
episodeNum: extractEpisodeNum(prog['episode-num'])
};
programmes.push(programme);
}
saxStream.on('opentag', function (node) {
currentTag = node.name;
const attr = node.attributes;
return { channels, programmes };
}
if (currentTag === 'channel') {
currentObject = {
id: attr.id,
name: null, // Will be populated by display-name tag
icon: null,
url: null
};
} else if (currentTag === 'programme') {
currentObject = {
channelId: attr.channel,
start: parseXmltvDate(attr.start),
stop: parseXmltvDate(attr.stop),
title: null,
subtitle: null,
description: null,
category: [],
icon: null,
date: null,
episodeNum: null
};
} else if (currentTag === 'icon') {
if (currentObject) {
currentObject.icon = attr.src;
}
}
textBuffer = '';
});
/**
* Extract text from XMLTV element (handles both string and object formats)
*/
function extractText(element) {
if (!element) return null;
if (typeof element === 'string') return element;
if (Array.isArray(element)) {
// Prefer English or first item
const en = element.find(e => e.lang === 'en' || !e.lang);
return extractText(en || element[0]);
}
if (element._) return element._;
if (element['#text']) return element['#text'];
return String(element);
}
saxStream.on('text', function (text) {
textBuffer += text;
});
/**
* Extract categories array
*/
function extractCategories(category) {
if (!category) return [];
const cats = Array.isArray(category) ? category : [category];
return cats.map(c => extractText(c)).filter(Boolean);
}
saxStream.on('cdata', function (text) {
textBuffer += text;
});
/**
* Extract episode number
*/
function extractEpisodeNum(episodeNum) {
if (!episodeNum) return null;
const nums = Array.isArray(episodeNum) ? episodeNum : [episodeNum];
saxStream.on('closetag', function (tagName) {
if (tagName === 'channel') {
if (currentObject) channels.push(currentObject);
currentObject = null;
} else if (tagName === 'programme') {
if (currentObject) programmes.push(currentObject);
currentObject = null;
} else if (currentObject) {
// Handle properties within objects
switch (tagName) {
case 'display-name': // channel name
if (!currentObject.name) currentObject.name = textBuffer;
break;
case 'url': // channel url
currentObject.url = textBuffer;
break;
case 'title':
currentObject.title = textBuffer;
break;
case 'sub-title':
currentObject.subtitle = textBuffer;
break;
case 'desc':
currentObject.description = textBuffer;
break;
case 'category':
if (textBuffer) currentObject.category.push(textBuffer);
break;
case 'date':
currentObject.date = textBuffer;
break;
case 'episode-num':
// Prefer system "xmltv_ns" or just take text
// Complex episode parsing logic can go here if needed
currentObject.episodeNum = textBuffer;
break;
}
}
});
for (const num of nums) {
if (typeof num === 'string') return num;
if (num._ || num['#text']) {
return num._ || num['#text'];
saxStream.on('end', function () {
resolve({ channels, programmes });
});
// Handle input type
if (typeof input === 'string') {
const inputStream = Readable.from([input]);
inputStream.pipe(saxStream);
} else {
input.pipe(saxStream);
}
}
return null;
});
}
/**
@@ -167,28 +188,40 @@ async function fetchAndParse(url) {
throw new Error(`Failed to fetch EPG: ${response.status} ${response.statusText}`);
}
let content;
const buffer = await response.arrayBuffer();
const bytes = new Uint8Array(buffer);
// Check for gzip magic bytes (1f 8b) to detect actual gzip content
// This handles cases where the server auto-decompresses or the URL doesn't reflect actual encoding
const isActuallyGzipped = bytes.length >= 2 && bytes[0] === 0x1f && bytes[1] === 0x8b;
if (isActuallyGzipped) {
// Handle gzipped EPG files (.xml.gz)
try {
const decompressed = await gunzip(Buffer.from(buffer));
content = decompressed.toString('utf-8');
} catch (err) {
throw new Error(`Failed to decompress gzipped EPG: ${err.message}`);
}
let stream;
if (response.body && typeof response.body.pipe === 'function') {
stream = response.body;
} else if (response.body) {
stream = Readable.fromWeb(response.body);
} else {
// Already plain text (or auto-decompressed by fetch)
content = Buffer.from(buffer).toString('utf-8');
stream = Readable.from([]);
}
return parse(content);
// Check for GZIP
// Note: We can't easily check for magic bytes on a stream without buffering.
// We'll rely on response headers or file extension mostly, or try to peek.
// For now, let's assume if content-encoding is gzip OR url ends in .gz
// However, undici/fetch usually handles 'Content-Encoding: gzip' automatically transparently.
// We only need to manually gunzip if the server serves it as application/octet-stream but it's actually gzipped,
// or if it's a .gz file download.
// A robust way for streams is checking magic bytes, but that requires peeking.
// Simplified approach: try to pipe through gunzip if the URL indicates it.
const isGzipped = url.endsWith('.gz') || (response.headers.get('content-type') || '').includes('gzip');
if (isGzipped) {
const gunzip = zlib.createGunzip();
stream.pipe(gunzip);
return parse(gunzip);
}
// In the previous version we read magic bytes.
// To support that with streams we'd need a peek stream.
// For now let's trust the transparent decompression of fetch or the URL.
return parse(stream);
}
module.exports = {
+88 -67
View File
@@ -1,8 +1,11 @@
/**
* M3U Playlist Parser
* Parses EXTM3U format playlists and extracts channel information
* M3U Playlist Parser (Streaming)
* Parses EXTM3U format playlists and extracts channel information line-by-line
*/
const readline = require('readline');
const { Readable } = require('stream');
/**
* Generate a simple stable ID from name and group
* @param {string} name - Channel name
@@ -21,69 +24,6 @@ function generateStableId(name, group) {
return `m3u_${Math.abs(hash).toString(36)}`;
}
/**
* Parse M3U playlist content
* @param {string} content - Raw M3U playlist content
* @returns {{ channels: Array, groups: Array }}
*/
function parse(content) {
const lines = content.split('\n').map(line => line.trim());
const channels = [];
const groupsSet = new Set();
// Verify it's a valid M3U file
if (!lines[0] || !lines[0].startsWith('#EXTM3U')) {
throw new Error('Invalid M3U format: missing #EXTM3U header');
}
let currentInfo = null;
let currentGroup = null;
for (let i = 1; i < lines.length; i++) {
const line = lines[i];
if (line.startsWith('#EXTINF:')) {
// Parse EXTINF line
currentInfo = parseExtinf(line);
if (currentInfo.groupTitle) {
groupsSet.add(currentInfo.groupTitle);
currentGroup = currentInfo.groupTitle;
}
} else if (line.startsWith('#EXTGRP:')) {
// Parse EXTGRP line (alternative group specification)
currentGroup = line.substring(8).trim();
groupsSet.add(currentGroup);
if (currentInfo) {
currentInfo.groupTitle = currentGroup;
}
} else if (line && !line.startsWith('#')) {
// This is a stream URL
if (currentInfo) {
const groupTitle = currentInfo.groupTitle || currentGroup || 'Uncategorized';
// Generate a stable ID: use tvgId if present, otherwise hash name+group
const stableId = currentInfo.tvgId || generateStableId(currentInfo.name, groupTitle);
channels.push({
...currentInfo,
id: stableId,
url: line,
groupTitle: groupTitle
});
currentInfo = null;
}
}
}
// Convert groups to array of objects
const groups = Array.from(groupsSet).map((name, index) => ({
id: `group_${index}`,
name,
channelCount: channels.filter(c => c.groupTitle === name).length
}));
return { channels, groups };
}
/**
* Parse EXTINF line and extract attributes
* @param {string} line - EXTINF line
@@ -138,6 +78,75 @@ function parseExtinf(line) {
return info;
}
/**
* Parse M3U content (Stream or String)
* @param {Readable|string} input - M3U content as Stream or String
* @returns {Promise<{ channels: Array, groups: Array }>}
*/
async function parse(input) {
const channels = [];
const groupsSet = new Set();
let currentInfo = null;
let currentGroup = null;
let inputStream;
if (typeof input === 'string') {
inputStream = Readable.from([input]);
} else {
inputStream = input;
}
const rl = readline.createInterface({
input: inputStream,
crlfDelay: Infinity
});
for await (const line of rl) {
const trimmed = line.trim();
if (!trimmed) continue;
if (trimmed.startsWith('#EXTINF:')) {
// Parse EXTINF line
currentInfo = parseExtinf(trimmed);
if (currentInfo.groupTitle) {
groupsSet.add(currentInfo.groupTitle);
currentGroup = currentInfo.groupTitle;
}
} else if (trimmed.startsWith('#EXTGRP:')) {
// Parse EXTGRP line (alternative group specification)
currentGroup = trimmed.substring(8).trim();
groupsSet.add(currentGroup);
if (currentInfo) {
currentInfo.groupTitle = currentGroup;
}
} else if (!trimmed.startsWith('#')) {
// This is a stream URL
if (currentInfo) {
const groupTitle = currentInfo.groupTitle || currentGroup || 'Uncategorized';
// Generate a stable ID: use tvgId if present, otherwise hash name+group
const stableId = currentInfo.tvgId || generateStableId(currentInfo.name, groupTitle);
channels.push({
...currentInfo,
id: stableId,
url: trimmed,
groupTitle: groupTitle
});
currentInfo = null;
}
}
}
// Convert groups to array of objects
const groups = Array.from(groupsSet).map((name, index) => ({
id: `group_${index}`,
name,
channelCount: channels.filter(c => c.groupTitle === name).length
}));
return { channels, groups };
}
/**
* Fetch and parse M3U from URL
* @param {string} url - M3U playlist URL
@@ -148,8 +157,20 @@ async function fetchAndParse(url) {
if (!response.ok) {
throw new Error(`Failed to fetch M3U: ${response.status} ${response.statusText}`);
}
const content = await response.text();
return parse(content);
// Check if body is a Node.js stream (undici/node-fetch) or web stream
let stream;
if (response.body && typeof response.body.pipe === 'function') {
stream = response.body;
} else if (response.body) {
// Convert Web Stream to Node Readable for readline
stream = Readable.fromWeb(response.body);
} else {
// Fallback for empty body
stream = Readable.from([]);
}
return parse(stream);
}
module.exports = { parse, parseExtinf, fetchAndParse };