Files

243 lines
6.3 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""
IPTV采集器主脚本
采集公开IPTV源,解析频道信息,生成M3U播放列表
"""
import os
import re
import sys
import time
import argparse
import logging
from urllib.parse import urlparse
import requests
import config
# 配置日志
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s [%(levelname)s] %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
)
logger = logging.getLogger(__name__)
def fetch_source(url):
"""下载源内容"""
headers = {"User-Agent": config.USER_AGENT}
try:
resp = requests.get(url, headers=headers, timeout=config.REQUEST_TIMEOUT)
resp.raise_for_status()
return resp.text
except Exception as e:
logger.warning("下载失败 %s: %s", url, e)
return None
def parse_m3u(content):
"""解析M3U格式内容,返回频道列表"""
channels = []
if not content:
return channels
lines = content.splitlines()
i = 0
while i < len(lines):
line = lines[i].strip()
if line.startswith("#EXTINF"):
# 解析属性
name = ""
group = ""
logo = ""
tvg_id = ""
# 提取逗号后的频道名
if "," in line:
name = line.split(",", 1)[1].strip()
# 提取属性
attrs = re.findall(r'(\w+-?\w*)="([^"]*)"', line)
for k, v in attrs:
if k == "group-title":
group = v
elif k == "tvg-logo":
logo = v
elif k == "tvg-id":
tvg_id = v
# 下一行是URL
if i + 1 < len(lines):
url_line = lines[i + 1].strip()
if url_line and not url_line.startswith("#"):
channels.append({
"name": name,
"group": group,
"logo": logo,
"tvg_id": tvg_id,
"url": url_line,
})
i += 1
i += 1
return channels
def parse_txt(content):
"""解析TXT格式内容(每行:频道名,URL)"""
channels = []
if not content:
return channels
for line in content.splitlines():
line = line.strip()
if not line or line.startswith("#"):
continue
if "," in line:
parts = line.split(",", 1)
name = parts[0].strip()
url = parts[1].strip()
if url.startswith("http"):
channels.append({
"name": name,
"group": "",
"logo": "",
"tvg_id": "",
"url": url,
})
return channels
def classify_channel(channel):
"""根据频道名和分组信息分类"""
name = channel.get("name", "")
group = channel.get("group", "")
text = f"{name} {group}"
for grp, keywords in config.GROUP_KEYWORDS.items():
for kw in keywords:
if kw.lower() in text.lower():
return grp
return config.DEFAULT_GROUP
def is_valid_url(url):
"""简单校验URL"""
try:
p = urlparse(url)
return p.scheme in ("http", "https") and bool(p.netloc)
except Exception:
return False
def collect_all():
"""采集所有源并合并去重"""
all_channels = []
seen_urls = set()
for src in config.SOURCES:
logger.info("采集源: %s", src)
content = fetch_source(src)
if not content:
continue
# 根据内容判断格式
if "#EXTM3U" in content or "#EXTINF" in content:
channels = parse_m3u(content)
else:
channels = parse_txt(content)
logger.info(" 解析到 %d 个频道", len(channels))
for ch in channels:
url = ch.get("url", "")
if not url or not is_valid_url(url):
continue
if url in seen_urls:
continue
seen_urls.add(url)
ch["group"] = classify_channel(ch)
all_channels.append(ch)
logger.info("合计有效频道: %d", len(all_channels))
return all_channels
def generate_m3u(channels, output_file):
"""生成M3U文件"""
os.makedirs(os.path.dirname(output_file), exist_ok=True)
# 按分组排序
group_order = ["央视", "卫视", "地方台", config.DEFAULT_GROUP]
channels.sort(key=lambda c: (
group_order.index(c["group"]) if c["group"] in group_order else len(group_order),
c["name"],
))
with open(output_file, "w", encoding="utf-8") as f:
f.write("#EXTM3U\n")
for ch in channels:
name = ch.get("name", "未知频道")
group = ch.get("group", config.DEFAULT_GROUP)
logo = ch.get("logo", "")
tvg_id = ch.get("tvg_id", "")
url = ch.get("url", "")
attrs = f'tvg-name="{name}" group-title="{group}"'
if logo:
attrs += f' tvg-logo="{logo}"'
if tvg_id:
attrs += f' tvg-id="{tvg_id}"'
f.write(f"#EXTINF:-1 {attrs},{name}\n")
f.write(f"{url}\n")
logger.info("已生成: %s", output_file)
def run_once():
"""执行一次采集"""
channels = collect_all()
if not channels:
logger.error("未采集到任何频道")
return False
generate_m3u(channels, config.OUTPUT_FILE)
return True
def run_daemon():
"""以守护模式运行,每天更新一次"""
try:
import schedule
except ImportError:
logger.error("守护模式需要安装 schedule: pip install schedule")
sys.exit(1)
logger.info("启动守护模式,每天 02:00 更新")
schedule.every().day.at("02:00").do(run_once)
# 启动时先执行一次
run_once()
while True:
schedule.run_pending()
time.sleep(60)
def main():
parser = argparse.ArgumentParser(description="IPTV采集器")
parser.add_argument("--daemon", action="store_true", help="以守护模式运行,每天自动更新")
args = parser.parse_args()
if args.daemon:
run_daemon()
else:
ok = run_once()
sys.exit(0 if ok else 1)
if __name__ == "__main__":
main()