AI 生成: IPTV直播源自动采集、检测、去重工具,支持从多个公开源采集直播源,自动检测可用性,去重后输出干净的M3U播放列表。

This commit is contained in:
admin
2026-10-03 08:31:56 +08:00
commit 90d1272210
9 changed files with 781 additions and 0 deletions
+86
View File
@@ -0,0 +1,86 @@
# IPTV 直播源自动采集检测去重
一个实用的 IPTV 直播源工具,自动从多个公开源采集直播源,检测可用性,去重后生成干净的 M3U 播放列表。
## 功能特性
- **自动采集**:从多个公开 IPTV 源采集直播源
- **并发检测**:多线程并发检测直播源可用性
- **智能去重**:基于 URL 和频道名称去重
- **分类整理**:自动按频道类型分类
- **M3U 输出**:生成标准 M3U 播放列表
- **Web 界面**:提供简单的 Web 界面查看结果
## 目录结构
```
iptv-source-collector/
├── README.md
├── requirements.txt
├── config.py # 配置文件
├── collector.py # 采集模块
├── checker.py # 检测模块
├── deduplicator.py # 去重模块
├── main.py # 主程序入口
├── app.py # Flask Web 应用
├── templates/
│ └── index.html # Web 界面
└── output/ # 输出目录(自动创建)
```
## 安装依赖
```bash
pip install -r requirements.txt
```
## 使用方法
### 命令行模式
```bash
# 完整流程:采集 -> 检测 -> 去重 -> 输出
python main.py
# 只采集不检测(快速获取所有源)
python main.py --no-check
# 指定输出文件
python main.py --output my_playlist.m3u
# 设置并发数
python main.py --threads 50
```
### Web 界面模式
```bash
python app.py
```
然后访问 http://127.0.0.1:5000 查看结果。
## 配置说明
编辑 `config.py` 可以自定义:
- `SOURCE_URLS`:采集源地址列表
- `CHECK_TIMEOUT`:检测超时时间(秒)
- `MAX_WORKERS`:并发线程数
- `MIN_SPEED`:最低下载速度要求(KB/s)
## 输出格式
生成的 M3U 文件格式:
```
#EXTM3U
#EXTINF:-1 tvg-name="CCTV-1" group-title="央视",CCTV-1
http://example.com/cctv1.m3u8
```
## 注意事项
- 本工具仅供学习和研究使用
- 请遵守相关法律法规,不要用于非法用途
- 采集源均来自公开网络,如有侵权请联系删除
+115
View File
@@ -0,0 +1,115 @@
# -*- coding: utf-8 -*-
"""Flask Web 应用:展示 IPTV 直播源结果"""
import os
import json
from flask import Flask, render_template, jsonify, request
from collector import collect_all
from checker import check_all
from deduplicator import deduplicate, group_by_category
from config import OUTPUT_DIR, OUTPUT_FILE
app = Flask(__name__)
# 缓存结果
_cache = {
"channels": [],
"groups": {},
"checked": False,
}
def load_from_file(filepath):
"""从 M3U 文件加载频道"""
channels = []
if not os.path.exists(filepath):
return channels
with open(filepath, "r", encoding="utf-8") as f:
current = None
for line in f:
line = line.strip()
if line.startswith("#EXTINF"):
name = line.split(",", 1)[1] if "," in line else ""
import re
tvg = re.search(r'tvg-name="([^"]*)"', line)
grp = re.search(r'group-title="([^"]*)"', line)
current = {
"name": name,
"tvg_name": tvg.group(1) if tvg else name,
"group": grp.group(1) if grp else "其他",
"url": "",
}
elif line and not line.startswith("#") and current:
current["url"] = line
channels.append(current)
current = None
return channels
@app.route("/")
def index():
"""首页"""
return render_template("index.html")
@app.route("/api/channels")
def api_channels():
"""获取频道列表"""
return jsonify({
"total": len(_cache["channels"]),
"channels": _cache["channels"],
})
@app.route("/api/groups")
def api_groups():
"""获取分组统计"""
groups = {}
for g, items in _cache["groups"].items():
groups[g] = len(items)
return jsonify(groups)
@app.route("/api/collect", methods=["POST"])
def api_collect():
"""触发采集+检测"""
do_check = request.json.get("check", True) if request.is_json else True
channels = collect_all()
if do_check:
channels = check_all(channels)
channels = deduplicate(channels)
_cache["channels"] = channels
_cache["groups"] = group_by_category(channels)
_cache["checked"] = do_check
return jsonify({
"success": True,
"total": len(channels),
})
@app.route("/api/load")
def api_load():
"""从输出文件加载"""
filepath = os.path.join(OUTPUT_DIR, OUTPUT_FILE)
channels = load_from_file(filepath)
_cache["channels"] = channels
_cache["groups"] = group_by_category(channels)
return jsonify({
"success": True,
"total": len(channels),
})
if __name__ == "__main__":
# 启动时尝试加载已有结果
filepath = os.path.join(OUTPUT_DIR, OUTPUT_FILE)
if os.path.exists(filepath):
channels = load_from_file(filepath)
_cache["channels"] = channels
_cache["groups"] = group_by_category(channels)
print(f"已加载 {len(channels)} 个频道")
app.run(host="0.0.0.0", port=5000, debug=False)
+80
View File
@@ -0,0 +1,80 @@
# -*- coding: utf-8 -*-
"""检测模块:并发检测 IPTV 源可用性"""
import requests
from concurrent.futures import ThreadPoolExecutor, as_completed
from config import CHECK_TIMEOUT, MAX_WORKERS, HEADERS
def check_channel(channel):
"""检测单个频道是否可用
返回 (channel, is_ok) 元组
"""
url = channel.get("url", "")
if not url:
return channel, False
try:
# 使用流式请求,只读取少量数据
resp = requests.get(
url,
headers=HEADERS,
timeout=CHECK_TIMEOUT,
stream=True,
allow_redirects=True,
)
# 状态码检查
if resp.status_code != 200:
return channel, False
# 读取前 1KB 数据验证内容
content = b""
for chunk in resp.iter_content(chunk_size=1024):
content += chunk
if len(content) >= 1024:
break
# 检查是否为有效的流内容(m3u8 或 ts 流)
if content:
text = content[:200].decode("utf-8", errors="ignore")
# m3u8 文件或以 #EXTM3U 开头
if "#EXTM3U" in text or "#EXTINF" in text:
return channel, True
# ts 流通常以 0x47 开头
if content[:1] == b"\x47":
return channel, True
# 其他情况认为可用(有些流返回的是二进制数据)
if len(content) > 100:
return channel, True
return channel, False
except Exception:
return channel, False
def check_all(channels):
"""并发检测所有频道"""
print(f"[检测] 开始检测 {len(channels)} 个频道,并发数 {MAX_WORKERS}")
valid = []
total = len(channels)
done = 0
with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
futures = {executor.submit(check_channel, ch): ch for ch in channels}
for future in as_completed(futures):
done += 1
channel, ok = future.result()
if ok:
valid.append(channel)
if done % 50 == 0 or done == total:
print(f"[检测进度] {done}/{total},可用 {len(valid)}")
print(f"[检测完成] 可用频道 {len(valid)}/{total}")
return valid
if __name__ == "__main__":
from collector import collect_all
channels = collect_all()
valid = check_all(channels)
print(f"最终可用:{len(valid)}")
+88
View File
@@ -0,0 +1,88 @@
# -*- coding: utf-8 -*-
"""采集模块:从多个源采集 IPTV 直播源"""
import requests
import re
from config import SOURCE_URLS, HEADERS
def parse_m3u(content):
"""解析 M3U 内容,返回频道列表
每个频道为 dict: {name, url, group, tvg_name}
"""
channels = []
lines = content.splitlines()
current = None
for line in lines:
line = line.strip()
if not line:
continue
if line.startswith("#EXTINF"):
# 解析 EXTINF 行
name = ""
group = ""
tvg_name = ""
# 提取逗号后的频道名
if "," in line:
name = line.split(",", 1)[1].strip()
# 提取属性
tvg_match = re.search(r'tvg-name="([^"]*)"', line)
if tvg_match:
tvg_name = tvg_match.group(1)
group_match = re.search(r'group-title="([^"]*)"', line)
if group_match:
group = group_match.group(1)
current = {
"name": name,
"url": "",
"group": group,
"tvg_name": tvg_name or name,
}
elif line.startswith("#"):
# 其他注释行忽略
continue
else:
# URL 行
if current is not None:
current["url"] = line
channels.append(current)
current = None
return channels
def collect_from_url(url):
"""从单个 URL 采集源"""
try:
resp = requests.get(url, headers=HEADERS, timeout=15)
resp.raise_for_status()
# 尝试自动识别编码
resp.encoding = resp.apparent_encoding or "utf-8"
channels = parse_m3u(resp.text)
print(f"[采集] {url} -> {len(channels)} 个频道")
return channels
except Exception as e:
print(f"[采集失败] {url}: {e}")
return []
def collect_all():
"""从所有源采集"""
all_channels = []
for url in SOURCE_URLS:
channels = collect_from_url(url)
all_channels.extend(channels)
print(f"[采集完成] 共采集 {len(all_channels)} 个频道")
return all_channels
if __name__ == "__main__":
result = collect_all()
for ch in result[:5]:
print(ch)
+25
View File
@@ -0,0 +1,25 @@
# -*- coding: utf-8 -*-
"""配置文件"""
# 采集源地址列表(公开的 IPTV 源)
SOURCE_URLS = [
"https://raw.githubusercontent.com/iptv-org/iptv/master/streams/cn.m3u",
"https://raw.githubusercontent.com/iptv-org/iptv/master/streams/hk.m3u",
"https://raw.githubusercontent.com/iptv-org/iptv/master/streams/tw.m3u",
"https://raw.githubusercontent.com/iptv-org/iptv/master/streams/us.m3u",
]
# 检测配置
CHECK_TIMEOUT = 5 # 单个源检测超时时间(秒)
MAX_WORKERS = 30 # 并发线程数
MIN_SPEED = 10 # 最低下载速度要求(KB/s),0 表示不限制
# 输出配置
OUTPUT_DIR = "output"
OUTPUT_FILE = "playlist.m3u"
# 请求头
HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
+70
View File
@@ -0,0 +1,70 @@
# -*- coding: utf-8 -*-
"""去重模块:基于 URL 和频道名去重"""
from urllib.parse import urlparse
def normalize_url(url):
"""标准化 URL,用于去重比较"""
try:
parsed = urlparse(url)
# 去掉协议差异,统一小写域名
host = parsed.netloc.lower()
path = parsed.path.rstrip("/")
return f"{host}{path}"
except Exception:
return url
def deduplicate(channels):
"""去重:
1. 相同 URL 的只保留一个
2. 相同频道名 + 相同 URL 的只保留一个
"""
seen_urls = set()
seen_names = {}
result = []
for ch in channels:
url = ch.get("url", "")
name = ch.get("name", "").strip()
if not url:
continue
norm_url = normalize_url(url)
# URL 去重
if norm_url in seen_urls:
continue
# 频道名去重:同一频道名保留第一个(可扩展为保留速度最快的)
if name and name in seen_names:
continue
seen_urls.add(norm_url)
if name:
seen_names[name] = True
result.append(ch)
print(f"[去重] {len(channels)} -> {len(result)} 个频道")
return result
def group_by_category(channels):
"""按分组分类"""
groups = {}
for ch in channels:
group = ch.get("group", "") or "其他"
groups.setdefault(group, []).append(ch)
return groups
if __name__ == "__main__":
test = [
{"name": "CCTV-1", "url": "http://a.com/1.m3u8", "group": "央视"},
{"name": "CCTV-1", "url": "http://b.com/1.m3u8", "group": "央视"},
{"name": "CCTV-2", "url": "http://a.com/1.m3u8", "group": "央视"},
{"name": "CCTV-3", "url": "http://c.com/3.m3u8", "group": "央视"},
]
print(deduplicate(test))
+83
View File
@@ -0,0 +1,83 @@
# -*- coding: utf-8 -*-
"""主程序:IPTV 直播源采集、检测、去重"""
import os
import sys
import argparse
from collector import collect_all
from checker import check_all
from deduplicator import deduplicate, group_by_category
from config import OUTPUT_DIR, OUTPUT_FILE
def write_m3u(channels, filepath):
"""写入 M3U 文件"""
os.makedirs(os.path.dirname(filepath) or ".", exist_ok=True)
with open(filepath, "w", encoding="utf-8") as f:
f.write("#EXTM3U\n")
for ch in channels:
name = ch.get("name", "未知频道")
group = ch.get("group", "其他") or "其他"
tvg_name = ch.get("tvg_name", name)
url = ch.get("url", "")
f.write(
f'#EXTINF:-1 tvg-name="{tvg_name}" group-title="{group}",{name}\n'
)
f.write(f"{url}\n")
print(f"[输出] 已写入 {filepath},共 {len(channels)} 个频道")
def main():
parser = argparse.ArgumentParser(description="IPTV 直播源采集检测去重工具")
parser.add_argument("--no-check", action="store_true", help="跳过检测步骤")
parser.add_argument("--output", default=OUTPUT_FILE, help="输出文件名")
parser.add_argument("--threads", type=int, default=None, help="并发线程数")
args = parser.parse_args()
if args.threads:
import config
config.MAX_WORKERS = args.threads
print("=" * 50)
print("IPTV 直播源采集检测去重工具")
print("=" * 50)
# 1. 采集
print("\n[步骤 1/4] 采集直播源...")
channels = collect_all()
if not channels:
print("未采集到任何频道,退出")
sys.exit(1)
# 2. 检测
if args.no_check:
print("\n[步骤 2/4] 跳过检测")
valid = channels
else:
print("\n[步骤 2/4] 检测直播源可用性...")
valid = check_all(channels)
if not valid:
print("没有可用频道,退出")
sys.exit(1)
# 3. 去重
print("\n[步骤 3/4] 去重...")
unique = deduplicate(valid)
# 4. 输出
print("\n[步骤 4/4] 输出结果...")
output_path = os.path.join(OUTPUT_DIR, args.output)
write_m3u(unique, output_path)
# 统计
groups = group_by_category(unique)
print("\n分类统计:")
for g, items in sorted(groups.items(), key=lambda x: -len(x[1])):
print(f" {g}: {len(items)} 个频道")
print(f"\n完成!输出文件:{output_path}")
if __name__ == "__main__":
main()
+2
View File
@@ -0,0 +1,2 @@
requests>=2.28.0
flask>=2.3.0
+232
View File
@@ -0,0 +1,232 @@
<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>IPTV 直播源管理</title>
<style>
* { margin: 0; padding: 0; box-sizing: border-box; }
body {
font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif;
background: #f5f7fa;
color: #333;
padding: 20px;
}
.container { max-width: 1200px; margin: 0 auto; }
h1 { margin-bottom: 20px; color: #2c3e50; }
.toolbar {
display: flex;
gap: 10px;
margin-bottom: 20px;
flex-wrap: wrap;
}
button {
padding: 10px 20px;
border: none;
border-radius: 6px;
background: #3498db;
color: white;
cursor: pointer;
font-size: 14px;
transition: background 0.2s;
}
button:hover { background: #2980b9; }
button.secondary { background: #95a5a6; }
button.secondary:hover { background: #7f8c8d; }
button:disabled { background: #bdc3c7; cursor: not-allowed; }
.stats {
display: flex;
gap: 20px;
margin-bottom: 20px;
flex-wrap: wrap;
}
.stat-card {
background: white;
padding: 15px 25px;
border-radius: 8px;
box-shadow: 0 2px 4px rgba(0,0,0,0.05);
flex: 1;
min-width: 150px;
}
.stat-card .label { font-size: 12px; color: #7f8c8d; }
.stat-card .value { font-size: 24px; font-weight: bold; color: #2c3e50; }
.search-box {
margin-bottom: 15px;
}
.search-box input {
width: 100%;
padding: 10px 15px;
border: 1px solid #ddd;
border-radius: 6px;
font-size: 14px;
}
table {
width: 100%;
background: white;
border-radius: 8px;
overflow: hidden;
box-shadow: 0 2px 4px rgba(0,0,0,0.05);
border-collapse: collapse;
}
th, td {
padding: 12px 15px;
text-align: left;
border-bottom: 1px solid #eee;
font-size: 14px;
}
th { background: #f8f9fa; font-weight: 600; color: #555; }
tr:hover { background: #f8f9fa; }
td.url { color: #7f8c8d; font-size: 12px; word-break: break-all; }
.group-tag {
display: inline-block;
padding: 2px 8px;
background: #e8f4fd;
color: #2980b9;
border-radius: 4px;
font-size: 12px;
}
.empty { text-align: center; padding: 40px; color: #95a5a6; }
.loading { text-align: center; padding: 20px; color: #3498db; }
</style>
</head>
<body>
<div class="container">
<h1>📺 IPTV 直播源管理</h1>
<div class="toolbar">
<button id="btn-collect">🔄 采集并检测</button>
<button id="btn-collect-nocheck" class="secondary">⚡ 仅采集(不检测)</button>
<button id="btn-load" class="secondary">📂 加载已有结果</button>
</div>
<div class="stats">
<div class="stat-card">
<div class="label">总频道数</div>
<div class="value" id="total">0</div>
</div>
<div class="stat-card">
<div class="label">分组数</div>
<div class="value" id="groups">0</div>
</div>
</div>
<div class="search-box">
<input type="text" id="search" placeholder="搜索频道名称或分组...">
</div>
<div id="loading" class="loading" style="display:none;">
⏳ 正在处理,请稍候...(采集和检测可能需要几分钟)
</div>
<table id="table" style="display:none;">
<thead>
<tr>
<th>频道名</th>
<th>分组</th>
<th>URL</th>
</tr>
</thead>
<tbody id="tbody"></tbody>
</table>
<div id="empty" class="empty">暂无数据,请点击上方按钮开始采集</div>
</div>
<script>
let allChannels = [];
const $ = (id) => document.getElementById(id);
async function loadChannels() {
const resp = await fetch('/api/channels');
const data = await resp.json();
allChannels = data.channels || [];
render();
}
async function loadGroups() {
const resp = await fetch('/api/groups');
const data = await resp.json();
$('groups').textContent = Object.keys(data).length;
}
function render() {
const keyword = $('search').value.trim().toLowerCase();
const filtered = allChannels.filter(ch => {
if (!keyword) return true;
return (ch.name || '').toLowerCase().includes(keyword) ||
(ch.group || '').toLowerCase().includes(keyword);
});
$('total').textContent = allChannels.length;
const tbody = $('tbody');
tbody.innerHTML = '';
if (filtered.length === 0) {
$('table').style.display = 'none';
$('empty').style.display = 'block';
$('empty').textContent = allChannels.length === 0 ? '暂无数据,请点击上方按钮开始采集' : '没有匹配的频道';
return;
}
$('table').style.display = 'table';
$('empty').style.display = 'none';
// 只渲染前 500 条,避免卡顿
const show = filtered.slice(0, 500);
for (const ch of show) {
const tr = document.createElement('tr');
tr.innerHTML = `
<td>${escapeHtml(ch.name || '')}</td>
<td><span class="group-tag">${escapeHtml(ch.group || '其他')}</span></td>
<td class="url">${escapeHtml(ch.url || '')}</td>
`;
tbody.appendChild(tr);
}
}
function escapeHtml(s) {
return String(s).replace(/[&<>"']/g, c => ({
'&': '&amp;', '<': '&lt;', '>': '&gt;', '"': '&quot;', "'": '&#39;'
})[c]);
}
async function collect(check) {
$('loading').style.display = 'block';
$('btn-collect').disabled = true;
$('btn-collect-nocheck').disabled = true;
try {
const resp = await fetch('/api/collect', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ check })
});
const data = await resp.json();
if (data.success) {
await loadChannels();
await loadGroups();
}
} catch (e) {
alert('采集失败:' + e.message);
} finally {
$('loading').style.display = 'none';
$('btn-collect').disabled = false;
$('btn-collect-nocheck').disabled = false;
}
}
$('btn-collect').addEventListener('click', () => collect(true));
$('btn-collect-nocheck').addEventListener('click', () => collect(false));
$('btn-load').addEventListener('click', async () => {
await fetch('/api/load');
await loadChannels();
await loadGroups();
});
$('search').addEventListener('input', render);
// 初始化
loadChannels().then(loadGroups);
</script>
</body>
</html>