Files

71 lines
1.8 KiB
Python

# -*- coding: utf-8 -*-
"""去重模块:基于 URL 和频道名去重"""
from urllib.parse import urlparse
def normalize_url(url):
"""标准化 URL,用于去重比较"""
try:
parsed = urlparse(url)
# 去掉协议差异,统一小写域名
host = parsed.netloc.lower()
path = parsed.path.rstrip("/")
return f"{host}{path}"
except Exception:
return url
def deduplicate(channels):
"""去重:
1. 相同 URL 的只保留一个
2. 相同频道名 + 相同 URL 的只保留一个
"""
seen_urls = set()
seen_names = {}
result = []
for ch in channels:
url = ch.get("url", "")
name = ch.get("name", "").strip()
if not url:
continue
norm_url = normalize_url(url)
# URL 去重
if norm_url in seen_urls:
continue
# 频道名去重:同一频道名保留第一个(可扩展为保留速度最快的)
if name and name in seen_names:
continue
seen_urls.add(norm_url)
if name:
seen_names[name] = True
result.append(ch)
print(f"[去重] {len(channels)} -> {len(result)} 个频道")
return result
def group_by_category(channels):
"""按分组分类"""
groups = {}
for ch in channels:
group = ch.get("group", "") or "其他"
groups.setdefault(group, []).append(ch)
return groups
if __name__ == "__main__":
test = [
{"name": "CCTV-1", "url": "http://a.com/1.m3u8", "group": "央视"},
{"name": "CCTV-1", "url": "http://b.com/1.m3u8", "group": "央视"},
{"name": "CCTV-2", "url": "http://a.com/1.m3u8", "group": "央视"},
{"name": "CCTV-3", "url": "http://c.com/3.m3u8", "group": "央视"},
]
print(deduplicate(test))