89 lines
2.3 KiB
Python
89 lines
2.3 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""采集模块:从多个源采集 IPTV 直播源"""
|
|
|
|
import requests
|
|
import re
|
|
from config import SOURCE_URLS, HEADERS
|
|
|
|
|
|
def parse_m3u(content):
|
|
"""解析 M3U 内容,返回频道列表
|
|
每个频道为 dict: {name, url, group, tvg_name}
|
|
"""
|
|
channels = []
|
|
lines = content.splitlines()
|
|
current = None
|
|
|
|
for line in lines:
|
|
line = line.strip()
|
|
if not line:
|
|
continue
|
|
|
|
if line.startswith("#EXTINF"):
|
|
# 解析 EXTINF 行
|
|
name = ""
|
|
group = ""
|
|
tvg_name = ""
|
|
|
|
# 提取逗号后的频道名
|
|
if "," in line:
|
|
name = line.split(",", 1)[1].strip()
|
|
|
|
# 提取属性
|
|
tvg_match = re.search(r'tvg-name="([^"]*)"', line)
|
|
if tvg_match:
|
|
tvg_name = tvg_match.group(1)
|
|
|
|
group_match = re.search(r'group-title="([^"]*)"', line)
|
|
if group_match:
|
|
group = group_match.group(1)
|
|
|
|
current = {
|
|
"name": name,
|
|
"url": "",
|
|
"group": group,
|
|
"tvg_name": tvg_name or name,
|
|
}
|
|
elif line.startswith("#"):
|
|
# 其他注释行忽略
|
|
continue
|
|
else:
|
|
# URL 行
|
|
if current is not None:
|
|
current["url"] = line
|
|
channels.append(current)
|
|
current = None
|
|
|
|
return channels
|
|
|
|
|
|
def collect_from_url(url):
|
|
"""从单个 URL 采集源"""
|
|
try:
|
|
resp = requests.get(url, headers=HEADERS, timeout=15)
|
|
resp.raise_for_status()
|
|
# 尝试自动识别编码
|
|
resp.encoding = resp.apparent_encoding or "utf-8"
|
|
channels = parse_m3u(resp.text)
|
|
print(f"[采集] {url} -> {len(channels)} 个频道")
|
|
return channels
|
|
except Exception as e:
|
|
print(f"[采集失败] {url}: {e}")
|
|
return []
|
|
|
|
|
|
def collect_all():
|
|
"""从所有源采集"""
|
|
all_channels = []
|
|
for url in SOURCE_URLS:
|
|
channels = collect_from_url(url)
|
|
all_channels.extend(channels)
|
|
print(f"[采集完成] 共采集 {len(all_channels)} 个频道")
|
|
return all_channels
|
|
|
|
|
|
if __name__ == "__main__":
|
|
result = collect_all()
|
|
for ch in result[:5]:
|
|
print(ch)
|