commit 90d1272210bdcb5567f6b0ec1852a9b8b7771ed2 Author: admin Date: Sat Oct 3 08:31:56 2026 +0800 AI 生成: IPTV直播源自动采集、检测、去重工具,支持从多个公开源采集直播源,自动检测可用性,去重后输出干净的M3U播放列表。 diff --git a/README.md b/README.md new file mode 100644 index 0000000..8a57080 --- /dev/null +++ b/README.md @@ -0,0 +1,86 @@ +# IPTV 直播源自动采集检测去重 + +一个实用的 IPTV 直播源工具,自动从多个公开源采集直播源,检测可用性,去重后生成干净的 M3U 播放列表。 + +## 功能特性 + +- **自动采集**:从多个公开 IPTV 源采集直播源 +- **并发检测**:多线程并发检测直播源可用性 +- **智能去重**:基于 URL 和频道名称去重 +- **分类整理**:自动按频道类型分类 +- **M3U 输出**:生成标准 M3U 播放列表 +- **Web 界面**:提供简单的 Web 界面查看结果 + +## 目录结构 + +``` +iptv-source-collector/ +├── README.md +├── requirements.txt +├── config.py # 配置文件 +├── collector.py # 采集模块 +├── checker.py # 检测模块 +├── deduplicator.py # 去重模块 +├── main.py # 主程序入口 +├── app.py # Flask Web 应用 +├── templates/ +│ └── index.html # Web 界面 +└── output/ # 输出目录(自动创建) +``` + +## 安装依赖 + +```bash +pip install -r requirements.txt +``` + +## 使用方法 + +### 命令行模式 + +```bash +# 完整流程:采集 -> 检测 -> 去重 -> 输出 +python main.py + +# 只采集不检测(快速获取所有源) +python main.py --no-check + +# 指定输出文件 +python main.py --output my_playlist.m3u + +# 设置并发数 +python main.py --threads 50 +``` + +### Web 界面模式 + +```bash +python app.py +``` + +然后访问 http://127.0.0.1:5000 查看结果。 + +## 配置说明 + +编辑 `config.py` 可以自定义: + +- `SOURCE_URLS`:采集源地址列表 +- `CHECK_TIMEOUT`:检测超时时间(秒) +- `MAX_WORKERS`:并发线程数 +- `MIN_SPEED`:最低下载速度要求(KB/s) + +## 输出格式 + +生成的 M3U 文件格式: + +``` +#EXTM3U +#EXTINF:-1 tvg-name="CCTV-1" group-title="央视",CCTV-1 +http://example.com/cctv1.m3u8 +``` + +## 注意事项 + +- 本工具仅供学习和研究使用 +- 请遵守相关法律法规,不要用于非法用途 +- 采集源均来自公开网络,如有侵权请联系删除 diff --git a/app.py b/app.py new file mode 100644 index 0000000..1c63fb4 --- /dev/null +++ b/app.py @@ -0,0 +1,115 @@ +# -*- coding: utf-8 -*- +"""Flask Web 应用:展示 IPTV 直播源结果""" + +import os +import json +from flask import Flask, render_template, jsonify, request +from collector import collect_all +from checker import check_all +from deduplicator import deduplicate, group_by_category +from config import OUTPUT_DIR, OUTPUT_FILE + +app = Flask(__name__) + +# 缓存结果 +_cache = { + "channels": [], + "groups": {}, + "checked": False, +} + + +def load_from_file(filepath): + """从 M3U 文件加载频道""" + channels = [] + if not os.path.exists(filepath): + return channels + with open(filepath, "r", encoding="utf-8") as f: + current = None + for line in f: + line = line.strip() + if line.startswith("#EXTINF"): + name = line.split(",", 1)[1] if "," in line else "" + import re + tvg = re.search(r'tvg-name="([^"]*)"', line) + grp = re.search(r'group-title="([^"]*)"', line) + current = { + "name": name, + "tvg_name": tvg.group(1) if tvg else name, + "group": grp.group(1) if grp else "其他", + "url": "", + } + elif line and not line.startswith("#") and current: + current["url"] = line + channels.append(current) + current = None + return channels + + +@app.route("/") + def index(): + """首页""" + return render_template("index.html") + + +@app.route("/api/channels") + def api_channels(): + """获取频道列表""" + return jsonify({ + "total": len(_cache["channels"]), + "channels": _cache["channels"], + }) + + +@app.route("/api/groups") + def api_groups(): + """获取分组统计""" + groups = {} + for g, items in _cache["groups"].items(): + groups[g] = len(items) + return jsonify(groups) + + +@app.route("/api/collect", methods=["POST"]) + def api_collect(): + """触发采集+检测""" + do_check = request.json.get("check", True) if request.is_json else True + + channels = collect_all() + if do_check: + channels = check_all(channels) + channels = deduplicate(channels) + + _cache["channels"] = channels + _cache["groups"] = group_by_category(channels) + _cache["checked"] = do_check + + return jsonify({ + "success": True, + "total": len(channels), + }) + + +@app.route("/api/load") + def api_load(): + """从输出文件加载""" + filepath = os.path.join(OUTPUT_DIR, OUTPUT_FILE) + channels = load_from_file(filepath) + _cache["channels"] = channels + _cache["groups"] = group_by_category(channels) + return jsonify({ + "success": True, + "total": len(channels), + }) + + +if __name__ == "__main__": + # 启动时尝试加载已有结果 + filepath = os.path.join(OUTPUT_DIR, OUTPUT_FILE) + if os.path.exists(filepath): + channels = load_from_file(filepath) + _cache["channels"] = channels + _cache["groups"] = group_by_category(channels) + print(f"已加载 {len(channels)} 个频道") + + app.run(host="0.0.0.0", port=5000, debug=False) diff --git a/checker.py b/checker.py new file mode 100644 index 0000000..3fefa36 --- /dev/null +++ b/checker.py @@ -0,0 +1,80 @@ +# -*- coding: utf-8 -*- +"""检测模块:并发检测 IPTV 源可用性""" + +import requests +from concurrent.futures import ThreadPoolExecutor, as_completed +from config import CHECK_TIMEOUT, MAX_WORKERS, HEADERS + + +def check_channel(channel): + """检测单个频道是否可用 + 返回 (channel, is_ok) 元组 + """ + url = channel.get("url", "") + if not url: + return channel, False + + try: + # 使用流式请求,只读取少量数据 + resp = requests.get( + url, + headers=HEADERS, + timeout=CHECK_TIMEOUT, + stream=True, + allow_redirects=True, + ) + # 状态码检查 + if resp.status_code != 200: + return channel, False + + # 读取前 1KB 数据验证内容 + content = b"" + for chunk in resp.iter_content(chunk_size=1024): + content += chunk + if len(content) >= 1024: + break + + # 检查是否为有效的流内容(m3u8 或 ts 流) + if content: + text = content[:200].decode("utf-8", errors="ignore") + # m3u8 文件或以 #EXTM3U 开头 + if "#EXTM3U" in text or "#EXTINF" in text: + return channel, True + # ts 流通常以 0x47 开头 + if content[:1] == b"\x47": + return channel, True + # 其他情况认为可用(有些流返回的是二进制数据) + if len(content) > 100: + return channel, True + + return channel, False + except Exception: + return channel, False + + +def check_all(channels): + """并发检测所有频道""" + print(f"[检测] 开始检测 {len(channels)} 个频道,并发数 {MAX_WORKERS}") + valid = [] + total = len(channels) + done = 0 + + with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor: + futures = {executor.submit(check_channel, ch): ch for ch in channels} + for future in as_completed(futures): + done += 1 + channel, ok = future.result() + if ok: + valid.append(channel) + if done % 50 == 0 or done == total: + print(f"[检测进度] {done}/{total},可用 {len(valid)}") + + print(f"[检测完成] 可用频道 {len(valid)}/{total}") + return valid + + +if __name__ == "__main__": + from collector import collect_all + channels = collect_all() + valid = check_all(channels) + print(f"最终可用:{len(valid)}") diff --git a/collector.py b/collector.py new file mode 100644 index 0000000..341dbd2 --- /dev/null +++ b/collector.py @@ -0,0 +1,88 @@ +# -*- coding: utf-8 -*- +"""采集模块:从多个源采集 IPTV 直播源""" + +import requests +import re +from config import SOURCE_URLS, HEADERS + + +def parse_m3u(content): + """解析 M3U 内容,返回频道列表 + 每个频道为 dict: {name, url, group, tvg_name} + """ + channels = [] + lines = content.splitlines() + current = None + + for line in lines: + line = line.strip() + if not line: + continue + + if line.startswith("#EXTINF"): + # 解析 EXTINF 行 + name = "" + group = "" + tvg_name = "" + + # 提取逗号后的频道名 + if "," in line: + name = line.split(",", 1)[1].strip() + + # 提取属性 + tvg_match = re.search(r'tvg-name="([^"]*)"', line) + if tvg_match: + tvg_name = tvg_match.group(1) + + group_match = re.search(r'group-title="([^"]*)"', line) + if group_match: + group = group_match.group(1) + + current = { + "name": name, + "url": "", + "group": group, + "tvg_name": tvg_name or name, + } + elif line.startswith("#"): + # 其他注释行忽略 + continue + else: + # URL 行 + if current is not None: + current["url"] = line + channels.append(current) + current = None + + return channels + + +def collect_from_url(url): + """从单个 URL 采集源""" + try: + resp = requests.get(url, headers=HEADERS, timeout=15) + resp.raise_for_status() + # 尝试自动识别编码 + resp.encoding = resp.apparent_encoding or "utf-8" + channels = parse_m3u(resp.text) + print(f"[采集] {url} -> {len(channels)} 个频道") + return channels + except Exception as e: + print(f"[采集失败] {url}: {e}") + return [] + + +def collect_all(): + """从所有源采集""" + all_channels = [] + for url in SOURCE_URLS: + channels = collect_from_url(url) + all_channels.extend(channels) + print(f"[采集完成] 共采集 {len(all_channels)} 个频道") + return all_channels + + +if __name__ == "__main__": + result = collect_all() + for ch in result[:5]: + print(ch) diff --git a/config.py b/config.py new file mode 100644 index 0000000..0b39794 --- /dev/null +++ b/config.py @@ -0,0 +1,25 @@ +# -*- coding: utf-8 -*- +"""配置文件""" + +# 采集源地址列表(公开的 IPTV 源) +SOURCE_URLS = [ + "https://raw.githubusercontent.com/iptv-org/iptv/master/streams/cn.m3u", + "https://raw.githubusercontent.com/iptv-org/iptv/master/streams/hk.m3u", + "https://raw.githubusercontent.com/iptv-org/iptv/master/streams/tw.m3u", + "https://raw.githubusercontent.com/iptv-org/iptv/master/streams/us.m3u", +] + +# 检测配置 +CHECK_TIMEOUT = 5 # 单个源检测超时时间(秒) +MAX_WORKERS = 30 # 并发线程数 +MIN_SPEED = 10 # 最低下载速度要求(KB/s),0 表示不限制 + +# 输出配置 +OUTPUT_DIR = "output" +OUTPUT_FILE = "playlist.m3u" + +# 请求头 +HEADERS = { + "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" +} diff --git a/deduplicator.py b/deduplicator.py new file mode 100644 index 0000000..e52890d --- /dev/null +++ b/deduplicator.py @@ -0,0 +1,70 @@ +# -*- coding: utf-8 -*- +"""去重模块:基于 URL 和频道名去重""" + +from urllib.parse import urlparse + + +def normalize_url(url): + """标准化 URL,用于去重比较""" + try: + parsed = urlparse(url) + # 去掉协议差异,统一小写域名 + host = parsed.netloc.lower() + path = parsed.path.rstrip("/") + return f"{host}{path}" + except Exception: + return url + + +def deduplicate(channels): + """去重: + 1. 相同 URL 的只保留一个 + 2. 相同频道名 + 相同 URL 的只保留一个 + """ + seen_urls = set() + seen_names = {} + result = [] + + for ch in channels: + url = ch.get("url", "") + name = ch.get("name", "").strip() + + if not url: + continue + + norm_url = normalize_url(url) + + # URL 去重 + if norm_url in seen_urls: + continue + + # 频道名去重:同一频道名保留第一个(可扩展为保留速度最快的) + if name and name in seen_names: + continue + + seen_urls.add(norm_url) + if name: + seen_names[name] = True + result.append(ch) + + print(f"[去重] {len(channels)} -> {len(result)} 个频道") + return result + + +def group_by_category(channels): + """按分组分类""" + groups = {} + for ch in channels: + group = ch.get("group", "") or "其他" + groups.setdefault(group, []).append(ch) + return groups + + +if __name__ == "__main__": + test = [ + {"name": "CCTV-1", "url": "http://a.com/1.m3u8", "group": "央视"}, + {"name": "CCTV-1", "url": "http://b.com/1.m3u8", "group": "央视"}, + {"name": "CCTV-2", "url": "http://a.com/1.m3u8", "group": "央视"}, + {"name": "CCTV-3", "url": "http://c.com/3.m3u8", "group": "央视"}, + ] + print(deduplicate(test)) diff --git a/main.py b/main.py new file mode 100644 index 0000000..7447b41 --- /dev/null +++ b/main.py @@ -0,0 +1,83 @@ +# -*- coding: utf-8 -*- +"""主程序:IPTV 直播源采集、检测、去重""" + +import os +import sys +import argparse +from collector import collect_all +from checker import check_all +from deduplicator import deduplicate, group_by_category +from config import OUTPUT_DIR, OUTPUT_FILE + + +def write_m3u(channels, filepath): + """写入 M3U 文件""" + os.makedirs(os.path.dirname(filepath) or ".", exist_ok=True) + with open(filepath, "w", encoding="utf-8") as f: + f.write("#EXTM3U\n") + for ch in channels: + name = ch.get("name", "未知频道") + group = ch.get("group", "其他") or "其他" + tvg_name = ch.get("tvg_name", name) + url = ch.get("url", "") + f.write( + f'#EXTINF:-1 tvg-name="{tvg_name}" group-title="{group}",{name}\n' + ) + f.write(f"{url}\n") + print(f"[输出] 已写入 {filepath},共 {len(channels)} 个频道") + + +def main(): + parser = argparse.ArgumentParser(description="IPTV 直播源采集检测去重工具") + parser.add_argument("--no-check", action="store_true", help="跳过检测步骤") + parser.add_argument("--output", default=OUTPUT_FILE, help="输出文件名") + parser.add_argument("--threads", type=int, default=None, help="并发线程数") + args = parser.parse_args() + + if args.threads: + import config + config.MAX_WORKERS = args.threads + + print("=" * 50) + print("IPTV 直播源采集检测去重工具") + print("=" * 50) + + # 1. 采集 + print("\n[步骤 1/4] 采集直播源...") + channels = collect_all() + if not channels: + print("未采集到任何频道,退出") + sys.exit(1) + + # 2. 检测 + if args.no_check: + print("\n[步骤 2/4] 跳过检测") + valid = channels + else: + print("\n[步骤 2/4] 检测直播源可用性...") + valid = check_all(channels) + + if not valid: + print("没有可用频道,退出") + sys.exit(1) + + # 3. 去重 + print("\n[步骤 3/4] 去重...") + unique = deduplicate(valid) + + # 4. 输出 + print("\n[步骤 4/4] 输出结果...") + output_path = os.path.join(OUTPUT_DIR, args.output) + write_m3u(unique, output_path) + + # 统计 + groups = group_by_category(unique) + print("\n分类统计:") + for g, items in sorted(groups.items(), key=lambda x: -len(x[1])): + print(f" {g}: {len(items)} 个频道") + + print(f"\n完成!输出文件:{output_path}") + + +if __name__ == "__main__": + main() diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..6f8d78b --- /dev/null +++ b/requirements.txt @@ -0,0 +1,2 @@ +requests>=2.28.0 +flask>=2.3.0 diff --git a/templates/index.html b/templates/index.html new file mode 100644 index 0000000..4c9cb6b --- /dev/null +++ b/templates/index.html @@ -0,0 +1,232 @@ + + + + + + IPTV 直播源管理 + + + +
+

📺 IPTV 直播源管理

+ +
+ + + +
+ +
+
+
总频道数
+
0
+
+
+
分组数
+
0
+
+
+ + + + + + + + + + + + + + + + +
暂无数据,请点击上方按钮开始采集
+
+ + + +