From ecfc269b495289018fece151b0c1d73a21a72e52 Mon Sep 17 00:00:00 2001 From: admin Date: Mon, 28 Sep 2026 22:18:45 +0800 Subject: [PATCH] =?UTF-8?q?AI=20=E7=94=9F=E6=88=90:=20=E4=B8=80=E4=B8=AA?= =?UTF-8?q?=E8=87=AA=E5=8A=A8=E9=87=87=E9=9B=86IPTV=E9=A2=91=E9=81=93?= =?UTF-8?q?=EF=BC=88=E5=A4=AE=E8=A7=86=E3=80=81=E5=8D=AB=E8=A7=86=E3=80=81?= =?UTF-8?q?=E5=9C=B0=E6=96=B9=E5=8F=B0=EF=BC=89=E5=B9=B6=E7=94=9F=E6=88=90?= =?UTF-8?q?M3U=E6=92=AD=E6=94=BE=E5=88=97=E8=A1=A8=E7=9A=84Python=E5=B7=A5?= =?UTF-8?q?=E5=85=B7=EF=BC=8C=E6=94=AF=E6=8C=81=E6=AF=8F=E6=97=A5=E5=AE=9A?= =?UTF-8?q?=E6=97=B6=E6=9B=B4=E6=96=B0=E3=80=82?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- README.md | 83 ++++++++++++++++ collector.py | 242 +++++++++++++++++++++++++++++++++++++++++++++++ config.py | 30 ++++++ requirements.txt | 2 + 4 files changed, 357 insertions(+) create mode 100644 README.md create mode 100644 collector.py create mode 100644 config.py create mode 100644 requirements.txt diff --git a/README.md b/README.md new file mode 100644 index 0000000..ee5d6e4 --- /dev/null +++ b/README.md @@ -0,0 +1,83 @@ +# IPTV Collector + +自动采集央视、卫视、地方台等IPTV频道,生成标准的M3U播放列表,支持每日定时更新。 + +## 功能特性 + +- 从多个公开源采集频道信息 +- 自动分类:央视、卫视、地方台 +- 生成标准M3U格式播放列表 +- 支持每日自动更新(通过cron或内置调度) +- 简洁实用,易于扩展 + +## 目录结构 + +``` +iptv-collector/ +├── collector.py # 主采集脚本 +├── config.py # 配置文件 +├── requirements.txt # 依赖 +├── output/ # 输出目录(自动生成) +│ └── iptv.m3u # 生成的播放列表 +└── README.md +``` + +## 安装 + +```bash +pip install -r requirements.txt +``` + +## 使用 + +### 单次采集 + +```bash +python collector.py +``` + +生成的文件位于 `output/iptv.m3u`。 + +### 每日自动更新 + +#### 方式一:使用cron(Linux/macOS) + +```bash +# 每天凌晨2点执行 +0 2 * * * cd /path/to/iptv-collector && python collector.py >> collector.log 2>&1 +``` + +#### 方式二:使用内置调度(需安装schedule) + +```bash +python collector.py --daemon +``` + +### 自定义配置 + +编辑 `config.py` 可修改: +- 采集源URL +- 输出路径 +- 请求超时时间 +- 用户代理 + +## 输出格式 + +生成的M3U文件示例: + +``` +#EXTM3U +#EXTINF:-1 tvg-name="CCTV-1" group-title="央视",CCTV-1 综合 +http://example.com/cctv1.m3u8 +... +``` + +## 注意事项 + +- 采集源为公开网络资源,请遵守相关法律法规 +- 部分源可能失效,脚本会自动过滤无效链接 +- 建议定期检查更新日志 + +## 许可证 + +MIT diff --git a/collector.py b/collector.py new file mode 100644 index 0000000..0a8635b --- /dev/null +++ b/collector.py @@ -0,0 +1,242 @@ +# -*- coding: utf-8 -*- +""" +IPTV采集器主脚本 +采集公开IPTV源,解析频道信息,生成M3U播放列表 +""" + +import os +import re +import sys +import time +import argparse +import logging +from urllib.parse import urlparse + +import requests + +import config + +# 配置日志 +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(message)s", + datefmt="%Y-%m-%d %H:%M:%S", +) +logger = logging.getLogger(__name__) + + +def fetch_source(url): + """下载源内容""" + headers = {"User-Agent": config.USER_AGENT} + try: + resp = requests.get(url, headers=headers, timeout=config.REQUEST_TIMEOUT) + resp.raise_for_status() + return resp.text + except Exception as e: + logger.warning("下载失败 %s: %s", url, e) + return None + + +def parse_m3u(content): + """解析M3U格式内容,返回频道列表""" + channels = [] + if not content: + return channels + + lines = content.splitlines() + i = 0 + while i < len(lines): + line = lines[i].strip() + if line.startswith("#EXTINF"): + # 解析属性 + name = "" + group = "" + logo = "" + tvg_id = "" + + # 提取逗号后的频道名 + if "," in line: + name = line.split(",", 1)[1].strip() + + # 提取属性 + attrs = re.findall(r'(\w+-?\w*)="([^"]*)"', line) + for k, v in attrs: + if k == "group-title": + group = v + elif k == "tvg-logo": + logo = v + elif k == "tvg-id": + tvg_id = v + + # 下一行是URL + if i + 1 < len(lines): + url_line = lines[i + 1].strip() + if url_line and not url_line.startswith("#"): + channels.append({ + "name": name, + "group": group, + "logo": logo, + "tvg_id": tvg_id, + "url": url_line, + }) + i += 1 + i += 1 + + return channels + + +def parse_txt(content): + """解析TXT格式内容(每行:频道名,URL)""" + channels = [] + if not content: + return channels + + for line in content.splitlines(): + line = line.strip() + if not line or line.startswith("#"): + continue + if "," in line: + parts = line.split(",", 1) + name = parts[0].strip() + url = parts[1].strip() + if url.startswith("http"): + channels.append({ + "name": name, + "group": "", + "logo": "", + "tvg_id": "", + "url": url, + }) + return channels + + +def classify_channel(channel): + """根据频道名和分组信息分类""" + name = channel.get("name", "") + group = channel.get("group", "") + text = f"{name} {group}" + + for grp, keywords in config.GROUP_KEYWORDS.items(): + for kw in keywords: + if kw.lower() in text.lower(): + return grp + return config.DEFAULT_GROUP + + +def is_valid_url(url): + """简单校验URL""" + try: + p = urlparse(url) + return p.scheme in ("http", "https") and bool(p.netloc) + except Exception: + return False + + +def collect_all(): + """采集所有源并合并去重""" + all_channels = [] + seen_urls = set() + + for src in config.SOURCES: + logger.info("采集源: %s", src) + content = fetch_source(src) + if not content: + continue + + # 根据内容判断格式 + if "#EXTM3U" in content or "#EXTINF" in content: + channels = parse_m3u(content) + else: + channels = parse_txt(content) + + logger.info(" 解析到 %d 个频道", len(channels)) + + for ch in channels: + url = ch.get("url", "") + if not url or not is_valid_url(url): + continue + if url in seen_urls: + continue + seen_urls.add(url) + ch["group"] = classify_channel(ch) + all_channels.append(ch) + + logger.info("合计有效频道: %d", len(all_channels)) + return all_channels + + +def generate_m3u(channels, output_file): + """生成M3U文件""" + os.makedirs(os.path.dirname(output_file), exist_ok=True) + + # 按分组排序 + group_order = ["央视", "卫视", "地方台", config.DEFAULT_GROUP] + channels.sort(key=lambda c: ( + group_order.index(c["group"]) if c["group"] in group_order else len(group_order), + c["name"], + )) + + with open(output_file, "w", encoding="utf-8") as f: + f.write("#EXTM3U\n") + for ch in channels: + name = ch.get("name", "未知频道") + group = ch.get("group", config.DEFAULT_GROUP) + logo = ch.get("logo", "") + tvg_id = ch.get("tvg_id", "") + url = ch.get("url", "") + + attrs = f'tvg-name="{name}" group-title="{group}"' + if logo: + attrs += f' tvg-logo="{logo}"' + if tvg_id: + attrs += f' tvg-id="{tvg_id}"' + + f.write(f"#EXTINF:-1 {attrs},{name}\n") + f.write(f"{url}\n") + + logger.info("已生成: %s", output_file) + + +def run_once(): + """执行一次采集""" + channels = collect_all() + if not channels: + logger.error("未采集到任何频道") + return False + generate_m3u(channels, config.OUTPUT_FILE) + return True + + +def run_daemon(): + """以守护模式运行,每天更新一次""" + try: + import schedule + except ImportError: + logger.error("守护模式需要安装 schedule: pip install schedule") + sys.exit(1) + + logger.info("启动守护模式,每天 02:00 更新") + schedule.every().day.at("02:00").do(run_once) + + # 启动时先执行一次 + run_once() + + while True: + schedule.run_pending() + time.sleep(60) + + +def main(): + parser = argparse.ArgumentParser(description="IPTV采集器") + parser.add_argument("--daemon", action="store_true", help="以守护模式运行,每天自动更新") + args = parser.parse_args() + + if args.daemon: + run_daemon() + else: + ok = run_once() + sys.exit(0 if ok else 1) + + +if __name__ == "__main__": + main() diff --git a/config.py b/config.py new file mode 100644 index 0000000..019a8fd --- /dev/null +++ b/config.py @@ -0,0 +1,30 @@ +# -*- coding: utf-8 -*- +"""配置文件""" + +# 采集源列表(公开的IPTV源) +# 每个源应返回M3U或TXT格式的频道列表 +SOURCES = [ + # 示例源,实际使用时请替换为可用的公开源 + "https://raw.githubusercontent.com/iptv-org/iptv/master/streams/cn.m3u", + "https://raw.githubusercontent.com/iptv-org/iptv/master/streams/cn_sichuan.m3u", + "https://raw.githubusercontent.com/iptv-org/iptv/master/streams/cn_guangdong.m3u", +] + +# 输出文件路径 +OUTPUT_FILE = "output/iptv.m3u" + +# 请求超时时间(秒) +REQUEST_TIMEOUT = 15 + +# 用户代理 +USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" + +# 频道分组关键词映射 +GROUP_KEYWORDS = { + "央视": ["CCTV", "央视", "CCTV-"], + "卫视": ["卫视", "TVS", "星空", "凤凰"], + "地方台": ["地方", "市", "县", "区", "广东", "四川", "湖南", "浙江", "江苏", "山东", "河南"], +} + +# 默认分组 +DEFAULT_GROUP = "其他" diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..8cde3ef --- /dev/null +++ b/requirements.txt @@ -0,0 +1,2 @@ +requests>=2.28.0 +schedule>=1.2.0