# -*- coding: utf-8 -*- """去重模块:基于 URL 和频道名去重""" from urllib.parse import urlparse def normalize_url(url): """标准化 URL,用于去重比较""" try: parsed = urlparse(url) # 去掉协议差异,统一小写域名 host = parsed.netloc.lower() path = parsed.path.rstrip("/") return f"{host}{path}" except Exception: return url def deduplicate(channels): """去重: 1. 相同 URL 的只保留一个 2. 相同频道名 + 相同 URL 的只保留一个 """ seen_urls = set() seen_names = {} result = [] for ch in channels: url = ch.get("url", "") name = ch.get("name", "").strip() if not url: continue norm_url = normalize_url(url) # URL 去重 if norm_url in seen_urls: continue # 频道名去重:同一频道名保留第一个(可扩展为保留速度最快的) if name and name in seen_names: continue seen_urls.add(norm_url) if name: seen_names[name] = True result.append(ch) print(f"[去重] {len(channels)} -> {len(result)} 个频道") return result def group_by_category(channels): """按分组分类""" groups = {} for ch in channels: group = ch.get("group", "") or "其他" groups.setdefault(group, []).append(ch) return groups if __name__ == "__main__": test = [ {"name": "CCTV-1", "url": "http://a.com/1.m3u8", "group": "央视"}, {"name": "CCTV-1", "url": "http://b.com/1.m3u8", "group": "央视"}, {"name": "CCTV-2", "url": "http://a.com/1.m3u8", "group": "央视"}, {"name": "CCTV-3", "url": "http://c.com/3.m3u8", "group": "央视"}, ] print(deduplicate(test))