AI 生成: AVTV - 自动收集、过滤、去重并生成 m3u 直播源订阅文件,每日自动更新,涵盖央视、卫视及地方台。

This commit is contained in:
admin
2026-09-29 05:40:04 +08:00
commit 2625d08d37
12 changed files with 507 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
__version__ = "1.0.0"
+36
View File
@@ -0,0 +1,36 @@
"""抓取远程直播源内容。"""
import logging
import requests
from .sources import REQUEST_TIMEOUT, USER_AGENT
log = logging.getLogger(__name__)
def fetch(url: str) -> str | None:
"""下载单个源的原始文本,失败返回 None。"""
try:
resp = requests.get(
url,
timeout=REQUEST_TIMEOUT,
headers={"User-Agent": USER_AGENT},
)
resp.raise_for_status()
resp.encoding = resp.apparent_encoding or "utf-8"
log.info("fetched %s (%d bytes)", url, len(resp.text))
return resp.text
except Exception as exc: # noqa: BLE001
log.warning("fetch failed %s: %s", url, exc)
return None
def fetch_all(urls: list[str]) -> list[str]:
"""批量抓取,返回成功的内容列表。"""
results: list[str] = []
for url in urls:
text = fetch(url)
if text:
results.append(text)
return results
+101
View File
@@ -0,0 +1,101 @@
"""频道过滤与去重。"""
import re
from urllib.parse import urlparse
from .parser import Channel
# 允许的协议
_ALLOWED_SCHEMES = {"http", "https", "rtmp", "rtsp", "rtp"}
# 无效/占位 URL 关键词
_BAD_URL_KEYWORDS = (
"example.com",
"localhost",
"127.0.0.1",
"0.0.0.0",
"test",
"demo",
)
# 需要排除的频道名关键词(成人、赌博等)
_BAD_NAME_KEYWORDS = (
"成人",
"色情",
"赌博",
"博彩",
"彩票",
"porn",
"xxx",
"adult",
"casino",
)
# 频道名规范化:去掉空格、标点、常见后缀
_NORMALIZE_RE = re.compile(r"[\s\-_·\|()()\[\]【】]+")
def _normalize_name(name: str) -> str:
"""频道名归一化,用于去重。"""
n = name.lower().strip()
# 去掉常见画质 / 后缀标记
for suffix in ("高清", "超清", "蓝光", "hd", "fhd", "4k", "8k", "sd"):
n = n.replace(suffix, "")
n = _NORMALIZE_RE.sub("", n)
return n
def is_valid(ch: Channel) -> bool:
"""判断频道是否有效。"""
if not ch.name or not ch.url:
return False
name_l = ch.name.lower()
url_l = ch.url.lower()
if any(k in name_l for k in _BAD_NAME_KEYWORDS):
return False
if any(k in url_l for k in _BAD_URL_KEYWORDS):
return False
try:
scheme = urlparse(ch.url).scheme.lower()
except Exception: # noqa: BLE001
return False
if scheme not in _ALLOWED_SCHEMES:
return False
return True
def dedupe(channels: list[Channel]) -> list[Channel]:
"""按频道名 + URL 去重,优先保留信息更完整的条目。"""
seen_names: set[str] = set()
seen_urls: set[str] = set()
result: list[Channel] = []
# 先按信息完整度排序(有 logo / group 的优先)
channels = sorted(
channels,
key=lambda c: (bool(c.logo), bool(c.group), bool(c.tvg_id)),
reverse=True,
)
for ch in channels:
key = _normalize_name(ch.name)
if not key:
continue
if key in seen_names:
continue
if ch.url in seen_urls:
continue
seen_names.add(key)
seen_urls.add(ch.url)
result.append(ch)
return result
def process(channels: list[Channel]) -> list[Channel]:
"""完整流程:过滤 + 去重。"""
valid = [c for c in channels if is_valid(c)]
return dedupe(valid)
+101
View File
@@ -0,0 +1,101 @@
"""生成 m3u 播放列表文件。"""
from pathlib import Path
from .parser import Channel
# 分类顺序
_GROUP_ORDER = ["央视", "卫视", "地方台", "港澳台", "其他"]
# 分类关键词
_CCTV_KEYWORDS = ("cctv", "cgtn", "央视", "中央")
_SAT_KEYWORDS = ("卫视", "satellite")
_LOCAL_KEYWORDS = (
"北京",
"上海",
"天津",
"重庆",
"广东",
"江苏",
"浙江",
"湖南",
"湖北",
"河南",
"河北",
"山东",
"山西",
"陕西",
"四川",
"云南",
"贵州",
"广西",
"福建",
"江西",
"安徽",
"辽宁",
"吉林",
"黑龙江",
"甘肃",
"青海",
"宁夏",
"新疆",
"西藏",
"内蒙古",
"海南",
"深圳",
"广州",
"杭州",
"南京",
"武汉",
"成都",
"西安",
)
_HMT_KEYWORDS = ("香港", "澳门", "台湾", "翡翠", "明珠", "凤凰", "tvb", "hkstv", "澳视", "莲花")
def _classify(ch: Channel) -> str:
"""根据频道名/分组归类。"""
text = f"{ch.group} {ch.name}".lower()
if any(k in text for k in _CCTV_KEYWORDS):
return "央视"
if any(k in text for k in _HMT_KEYWORDS):
return "港澳台"
if any(k in text for k in _SAT_KEYWORDS):
return "卫视"
if any(k in ch.name for k in _LOCAL_KEYWORDS):
return "地方台"
return "其他"
def build_m3u(channels: list[Channel]) -> str:
"""将频道列表转换为 m3u 字符串。"""
# 分组
grouped: dict[str, list[Channel]] = {g: [] for g in _GROUP_ORDER}
for ch in channels:
grouped[_classify(ch)].append(ch)
lines: list[str] = ["#EXTM3U"]
for group in _GROUP_ORDER:
items = grouped.get(group, [])
if not items:
continue
for ch in items:
attrs = []
if ch.tvg_id:
attrs.append(f'tvg-id="{ch.tvg_id}"')
attrs.append(f'tvg-name="{ch.name}"')
if ch.logo:
attrs.append(f'tvg-logo="{ch.logo}"')
attrs.append(f'group-title="{group}"')
attr_str = " ".join(attrs)
lines.append(f"#EXTINF:-1 {attr_str},{ch.name}")
lines.append(ch.url)
return "\n".join(lines) + "\n"
def write_m3u(channels: list[Channel], out_path: str | Path) -> Path:
"""写出 m3u 文件,自动创建目录。"""
out = Path(out_path)
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(build_m3u(channels), encoding="utf-8")
return out
+88
View File
@@ -0,0 +1,88 @@
"""解析 m3u / txt 格式的直播源文本,提取频道条目。"""
import re
from dataclasses import dataclass, field
@dataclass
class Channel:
"""单个频道条目。"""
name: str
url: str
logo: str = ""
group: str = ""
tvg_id: str = ""
extra: dict = field(default_factory=dict)
_ATTR_RE = re.compile(r'([\w-]+)="([^"]*)"')
def _parse_extinf(line: str) -> dict:
"""解析 #EXTINF 行中的属性。"""
attrs = {}
for key, value in _ATTR_RE.findall(line):
attrs[key.lower()] = value
# 逗号后面的名称
if "," in line:
attrs["name"] = line.rsplit(",", 1)[1].strip()
return attrs
def parse_m3u(text: str) -> list[Channel]:
"""解析 m3u 文本。"""
channels: list[Channel] = []
lines = [ln.strip() for ln in text.splitlines()]
i = 0
while i < len(lines):
line = lines[i]
if line.startswith("#EXTINF"):
attrs = _parse_extinf(line)
# 下一行非注释的即为 URL
j = i + 1
while j < len(lines) and (not lines[j] or lines[j].startswith("#")):
j += 1
if j < len(lines):
url = lines[j]
name = attrs.get("name") or attrs.get("tvg-name") or ""
if name and url:
channels.append(
Channel(
name=name,
url=url,
logo=attrs.get("tvg-logo", ""),
group=attrs.get("group-title", ""),
tvg_id=attrs.get("tvg-id", ""),
)
)
i = j + 1
continue
i += 1
return channels
def parse_txt(text: str) -> list[Channel]:
"""解析常见 txt 格式:频道名,URL 或 频道名 URL。"""
channels: list[Channel] = []
for raw in text.splitlines():
line = raw.strip()
if not line or line.startswith("#"):
continue
# 支持 , 或空格分隔
if "," in line:
name, url = line.split(",", 1)
elif " " in line:
name, url = line.split(" ", 1)
else:
continue
name, url = name.strip(), url.strip()
if name and url.startswith(("http://", "https://", "rtmp://", "rtsp://")):
channels.append(Channel(name=name, url=url))
return channels
def parse(text: str) -> list[Channel]:
"""自动识别格式并解析。"""
if "#EXTM3U" in text or "#EXTINF" in text:
return parse_m3u(text)
return parse_txt(text)
+25
View File
@@ -0,0 +1,25 @@
"""直播源地址列表。可自由增删。"""
# 公开的 m3u / txt 直播源地址
SOURCES = [
# iptv-org 中文频道
"https://iptv-org.github.io/iptv/languages/zho.m3u",
"https://iptv-org.github.io/iptv/countries/cn.m3u",
"https://iptv-org.github.io/iptv/countries/hk.m3u",
"https://iptv-org.github.io/iptv/countries/tw.m3u",
# YanG-1989 维护的国内源
"https://raw.githubusercontent.com/YanG-1989/m3u/main/Gather.m3u",
"https://raw.githubusercontent.com/YanG-1989/m3u/main/iptv.m3u",
# 其他常见公开源
"https://raw.githubusercontent.com/iptv-org/iptv/master/streams/cn.m3u",
]
# 请求超时(秒)
REQUEST_TIMEOUT = 15
# User-Agent
USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/122.0.0.0 Safari/537.36"
)