"""频道过滤与去重。""" import re from urllib.parse import urlparse from .parser import Channel # 允许的协议 _ALLOWED_SCHEMES = {"http", "https", "rtmp", "rtsp", "rtp"} # 无效/占位 URL 关键词 _BAD_URL_KEYWORDS = ( "example.com", "localhost", "127.0.0.1", "0.0.0.0", "test", "demo", ) # 需要排除的频道名关键词(成人、赌博等) _BAD_NAME_KEYWORDS = ( "成人", "色情", "赌博", "博彩", "彩票", "porn", "xxx", "adult", "casino", ) # 频道名规范化:去掉空格、标点、常见后缀 _NORMALIZE_RE = re.compile(r"[\s\-_·\|()()\[\]【】]+") def _normalize_name(name: str) -> str: """频道名归一化,用于去重。""" n = name.lower().strip() # 去掉常见画质 / 后缀标记 for suffix in ("高清", "超清", "蓝光", "hd", "fhd", "4k", "8k", "sd"): n = n.replace(suffix, "") n = _NORMALIZE_RE.sub("", n) return n def is_valid(ch: Channel) -> bool: """判断频道是否有效。""" if not ch.name or not ch.url: return False name_l = ch.name.lower() url_l = ch.url.lower() if any(k in name_l for k in _BAD_NAME_KEYWORDS): return False if any(k in url_l for k in _BAD_URL_KEYWORDS): return False try: scheme = urlparse(ch.url).scheme.lower() except Exception: # noqa: BLE001 return False if scheme not in _ALLOWED_SCHEMES: return False return True def dedupe(channels: list[Channel]) -> list[Channel]: """按频道名 + URL 去重,优先保留信息更完整的条目。""" seen_names: set[str] = set() seen_urls: set[str] = set() result: list[Channel] = [] # 先按信息完整度排序(有 logo / group 的优先) channels = sorted( channels, key=lambda c: (bool(c.logo), bool(c.group), bool(c.tvg_id)), reverse=True, ) for ch in channels: key = _normalize_name(ch.name) if not key: continue if key in seen_names: continue if ch.url in seen_urls: continue seen_names.add(key) seen_urls.add(ch.url) result.append(ch) return result def process(channels: list[Channel]) -> list[Channel]: """完整流程:过滤 + 去重。""" valid = [c for c in channels if is_valid(c)] return dedupe(valid)