Files

102 lines
2.4 KiB
Python

"""频道过滤与去重。"""
import re
from urllib.parse import urlparse
from .parser import Channel
# 允许的协议
_ALLOWED_SCHEMES = {"http", "https", "rtmp", "rtsp", "rtp"}
# 无效/占位 URL 关键词
_BAD_URL_KEYWORDS = (
"example.com",
"localhost",
"127.0.0.1",
"0.0.0.0",
"test",
"demo",
)
# 需要排除的频道名关键词(成人、赌博等)
_BAD_NAME_KEYWORDS = (
"成人",
"色情",
"赌博",
"博彩",
"彩票",
"porn",
"xxx",
"adult",
"casino",
)
# 频道名规范化:去掉空格、标点、常见后缀
_NORMALIZE_RE = re.compile(r"[\s\-_·\|()()\[\]【】]+")
def _normalize_name(name: str) -> str:
"""频道名归一化,用于去重。"""
n = name.lower().strip()
# 去掉常见画质 / 后缀标记
for suffix in ("高清", "超清", "蓝光", "hd", "fhd", "4k", "8k", "sd"):
n = n.replace(suffix, "")
n = _NORMALIZE_RE.sub("", n)
return n
def is_valid(ch: Channel) -> bool:
"""判断频道是否有效。"""
if not ch.name or not ch.url:
return False
name_l = ch.name.lower()
url_l = ch.url.lower()
if any(k in name_l for k in _BAD_NAME_KEYWORDS):
return False
if any(k in url_l for k in _BAD_URL_KEYWORDS):
return False
try:
scheme = urlparse(ch.url).scheme.lower()
except Exception: # noqa: BLE001
return False
if scheme not in _ALLOWED_SCHEMES:
return False
return True
def dedupe(channels: list[Channel]) -> list[Channel]:
"""按频道名 + URL 去重,优先保留信息更完整的条目。"""
seen_names: set[str] = set()
seen_urls: set[str] = set()
result: list[Channel] = []
# 先按信息完整度排序(有 logo / group 的优先)
channels = sorted(
channels,
key=lambda c: (bool(c.logo), bool(c.group), bool(c.tvg_id)),
reverse=True,
)
for ch in channels:
key = _normalize_name(ch.name)
if not key:
continue
if key in seen_names:
continue
if ch.url in seen_urls:
continue
seen_names.add(key)
seen_urls.add(ch.url)
result.append(ch)
return result
def process(channels: list[Channel]) -> list[Channel]:
"""完整流程:过滤 + 去重。"""
valid = [c for c in channels if is_valid(c)]
return dedupe(valid)