commit 2625d08d374827de36cc9e1c98534ed82b1ee810 Author: admin Date: Tue Sep 29 05:40:04 2026 +0800 AI 生成: AVTV - 自动收集、过滤、去重并生成 m3u 直播源订阅文件,每日自动更新,涵盖央视、卫视及地方台。 diff --git a/.github/workflows/update.yml b/.github/workflows/update.yml new file mode 100644 index 0000000..097f913 --- /dev/null +++ b/.github/workflows/update.yml @@ -0,0 +1,39 @@ +name: Update IPTV Playlist + +on: + schedule: + # 每天 UTC 02:00(北京时间 10:00) + - cron: "0 2 * * *" + workflow_dispatch: + +permissions: + contents: write + +jobs: + update: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.11" + + - name: Install dependencies + run: pip install -r requirements.txt + + - name: Run AVTV + run: python main.py + + - name: Commit and push + run: | + git config user.name "github-actions[bot]" + git config user.email "github-actions[bot]@users.noreply.github.com" + git add output/avtv.m3u + if git diff --cached --quiet; then + echo "No changes" + else + git commit -m "chore: update playlist $(date -u +%Y-%m-%d)" + git push + fi diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..242b13e --- /dev/null +++ b/.gitignore @@ -0,0 +1,5 @@ +__pycache__/ +*.pyc +.venv/ +venv/ +.env diff --git a/README.md b/README.md new file mode 100644 index 0000000..262153a --- /dev/null +++ b/README.md @@ -0,0 +1,59 @@ +# AVTV + +自动收集、过滤、去重并生成 m3u 格式直播源订阅文件。每天自动更新,覆盖央视、卫视及地方台。 + +## 功能特性 + +- 从多个公开直播源地址抓取频道列表 +- 智能过滤:去除无效链接、重复频道、非法协议 +- 自动去重:按频道名 + URL 双重去重 +- 分类整理:央视 / 卫视 / 地方台 / 其他 +- 生成标准 m3u 播放列表 +- 支持 GitHub Actions 每日自动更新 + +## 目录结构 + +``` +avtv/ +├── avtv/ +│ ├── __init__.py +│ ├── fetcher.py # 抓取直播源 +│ ├── parser.py # 解析 m3u / txt 源 +│ ├── filter.py # 过滤与去重 +│ ├── generator.py # 生成 m3u 文件 +│ └── sources.py # 直播源地址列表 +├── output/ +│ └── avtv.m3u # 生成的订阅文件 +├── main.py # 主入口 +├── requirements.txt +└── .github/workflows/update.yml +``` + +## 本地运行 + +```bash +pip install -r requirements.txt +python main.py +``` + +生成的订阅文件位于 `output/avtv.m3u`。 + +## 订阅地址 + +部署到 GitHub 后,可通过以下地址订阅: + +``` +https://raw.githubusercontent.com/<你的用户名>/<仓库名>/main/output/avtv.m3u +``` + +## 自动更新 + +项目内置 GitHub Actions 工作流,每天 UTC 02:00 自动运行并提交更新。 + +## 自定义直播源 + +编辑 `avtv/sources.py` 中的 `SOURCES` 列表即可添加或删除源。 + +## License + +MIT diff --git a/avtv/__init__.py b/avtv/__init__.py new file mode 100644 index 0000000..5becc17 --- /dev/null +++ b/avtv/__init__.py @@ -0,0 +1 @@ +__version__ = "1.0.0" diff --git a/avtv/fetcher.py b/avtv/fetcher.py new file mode 100644 index 0000000..28faeba --- /dev/null +++ b/avtv/fetcher.py @@ -0,0 +1,36 @@ +"""抓取远程直播源内容。""" + +import logging + +import requests + +from .sources import REQUEST_TIMEOUT, USER_AGENT + +log = logging.getLogger(__name__) + + +def fetch(url: str) -> str | None: + """下载单个源的原始文本,失败返回 None。""" + try: + resp = requests.get( + url, + timeout=REQUEST_TIMEOUT, + headers={"User-Agent": USER_AGENT}, + ) + resp.raise_for_status() + resp.encoding = resp.apparent_encoding or "utf-8" + log.info("fetched %s (%d bytes)", url, len(resp.text)) + return resp.text + except Exception as exc: # noqa: BLE001 + log.warning("fetch failed %s: %s", url, exc) + return None + + +def fetch_all(urls: list[str]) -> list[str]: + """批量抓取,返回成功的内容列表。""" + results: list[str] = [] + for url in urls: + text = fetch(url) + if text: + results.append(text) + return results diff --git a/avtv/filter.py b/avtv/filter.py new file mode 100644 index 0000000..df3f35b --- /dev/null +++ b/avtv/filter.py @@ -0,0 +1,101 @@ +"""频道过滤与去重。""" + +import re +from urllib.parse import urlparse + +from .parser import Channel + +# 允许的协议 +_ALLOWED_SCHEMES = {"http", "https", "rtmp", "rtsp", "rtp"} + +# 无效/占位 URL 关键词 +_BAD_URL_KEYWORDS = ( + "example.com", + "localhost", + "127.0.0.1", + "0.0.0.0", + "test", + "demo", +) + +# 需要排除的频道名关键词(成人、赌博等) +_BAD_NAME_KEYWORDS = ( + "成人", + "色情", + "赌博", + "博彩", + "彩票", + "porn", + "xxx", + "adult", + "casino", +) + +# 频道名规范化:去掉空格、标点、常见后缀 +_NORMALIZE_RE = re.compile(r"[\s\-_·\|()()\[\]【】]+") + + +def _normalize_name(name: str) -> str: + """频道名归一化,用于去重。""" + n = name.lower().strip() + # 去掉常见画质 / 后缀标记 + for suffix in ("高清", "超清", "蓝光", "hd", "fhd", "4k", "8k", "sd"): + n = n.replace(suffix, "") + n = _NORMALIZE_RE.sub("", n) + return n + + +def is_valid(ch: Channel) -> bool: + """判断频道是否有效。""" + if not ch.name or not ch.url: + return False + name_l = ch.name.lower() + url_l = ch.url.lower() + + if any(k in name_l for k in _BAD_NAME_KEYWORDS): + return False + if any(k in url_l for k in _BAD_URL_KEYWORDS): + return False + + try: + scheme = urlparse(ch.url).scheme.lower() + except Exception: # noqa: BLE001 + return False + if scheme not in _ALLOWED_SCHEMES: + return False + + return True + + +def dedupe(channels: list[Channel]) -> list[Channel]: + """按频道名 + URL 去重,优先保留信息更完整的条目。""" + seen_names: set[str] = set() + seen_urls: set[str] = set() + result: list[Channel] = [] + + # 先按信息完整度排序(有 logo / group 的优先) + channels = sorted( + channels, + key=lambda c: (bool(c.logo), bool(c.group), bool(c.tvg_id)), + reverse=True, + ) + + for ch in channels: + key = _normalize_name(ch.name) + if not key: + continue + if key in seen_names: + continue + if ch.url in seen_urls: + continue + seen_names.add(key) + seen_urls.add(ch.url) + result.append(ch) + + return result + + +def process(channels: list[Channel]) -> list[Channel]: + """完整流程:过滤 + 去重。""" + valid = [c for c in channels if is_valid(c)] + return dedupe(valid) diff --git a/avtv/generator.py b/avtv/generator.py new file mode 100644 index 0000000..2351533 --- /dev/null +++ b/avtv/generator.py @@ -0,0 +1,101 @@ +"""生成 m3u 播放列表文件。""" + +from pathlib import Path + +from .parser import Channel + +# 分类顺序 +_GROUP_ORDER = ["央视", "卫视", "地方台", "港澳台", "其他"] + +# 分类关键词 +_CCTV_KEYWORDS = ("cctv", "cgtn", "央视", "中央") +_SAT_KEYWORDS = ("卫视", "satellite") +_LOCAL_KEYWORDS = ( + "北京", + "上海", + "天津", + "重庆", + "广东", + "江苏", + "浙江", + "湖南", + "湖北", + "河南", + "河北", + "山东", + "山西", + "陕西", + "四川", + "云南", + "贵州", + "广西", + "福建", + "江西", + "安徽", + "辽宁", + "吉林", + "黑龙江", + "甘肃", + "青海", + "宁夏", + "新疆", + "西藏", + "内蒙古", + "海南", + "深圳", + "广州", + "杭州", + "南京", + "武汉", + "成都", + "西安", +) +_HMT_KEYWORDS = ("香港", "澳门", "台湾", "翡翠", "明珠", "凤凰", "tvb", "hkstv", "澳视", "莲花") + + +def _classify(ch: Channel) -> str: + """根据频道名/分组归类。""" + text = f"{ch.group} {ch.name}".lower() + if any(k in text for k in _CCTV_KEYWORDS): + return "央视" + if any(k in text for k in _HMT_KEYWORDS): + return "港澳台" + if any(k in text for k in _SAT_KEYWORDS): + return "卫视" + if any(k in ch.name for k in _LOCAL_KEYWORDS): + return "地方台" + return "其他" + + +def build_m3u(channels: list[Channel]) -> str: + """将频道列表转换为 m3u 字符串。""" + # 分组 + grouped: dict[str, list[Channel]] = {g: [] for g in _GROUP_ORDER} + for ch in channels: + grouped[_classify(ch)].append(ch) + + lines: list[str] = ["#EXTM3U"] + for group in _GROUP_ORDER: + items = grouped.get(group, []) + if not items: + continue + for ch in items: + attrs = [] + if ch.tvg_id: + attrs.append(f'tvg-id="{ch.tvg_id}"') + attrs.append(f'tvg-name="{ch.name}"') + if ch.logo: + attrs.append(f'tvg-logo="{ch.logo}"') + attrs.append(f'group-title="{group}"') + attr_str = " ".join(attrs) + lines.append(f"#EXTINF:-1 {attr_str},{ch.name}") + lines.append(ch.url) + return "\n".join(lines) + "\n" + + +def write_m3u(channels: list[Channel], out_path: str | Path) -> Path: + """写出 m3u 文件,自动创建目录。""" + out = Path(out_path) + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text(build_m3u(channels), encoding="utf-8") + return out diff --git a/avtv/parser.py b/avtv/parser.py new file mode 100644 index 0000000..dc3936b --- /dev/null +++ b/avtv/parser.py @@ -0,0 +1,88 @@ +"""解析 m3u / txt 格式的直播源文本,提取频道条目。""" + +import re +from dataclasses import dataclass, field + + +@dataclass +class Channel: + """单个频道条目。""" + name: str + url: str + logo: str = "" + group: str = "" + tvg_id: str = "" + extra: dict = field(default_factory=dict) + + +_ATTR_RE = re.compile(r'([\w-]+)="([^"]*)"') + + +def _parse_extinf(line: str) -> dict: + """解析 #EXTINF 行中的属性。""" + attrs = {} + for key, value in _ATTR_RE.findall(line): + attrs[key.lower()] = value + # 逗号后面的名称 + if "," in line: + attrs["name"] = line.rsplit(",", 1)[1].strip() + return attrs + + +def parse_m3u(text: str) -> list[Channel]: + """解析 m3u 文本。""" + channels: list[Channel] = [] + lines = [ln.strip() for ln in text.splitlines()] + i = 0 + while i < len(lines): + line = lines[i] + if line.startswith("#EXTINF"): + attrs = _parse_extinf(line) + # 下一行非注释的即为 URL + j = i + 1 + while j < len(lines) and (not lines[j] or lines[j].startswith("#")): + j += 1 + if j < len(lines): + url = lines[j] + name = attrs.get("name") or attrs.get("tvg-name") or "" + if name and url: + channels.append( + Channel( + name=name, + url=url, + logo=attrs.get("tvg-logo", ""), + group=attrs.get("group-title", ""), + tvg_id=attrs.get("tvg-id", ""), + ) + ) + i = j + 1 + continue + i += 1 + return channels + + +def parse_txt(text: str) -> list[Channel]: + """解析常见 txt 格式:频道名,URL 或 频道名 URL。""" + channels: list[Channel] = [] + for raw in text.splitlines(): + line = raw.strip() + if not line or line.startswith("#"): + continue + # 支持 , 或空格分隔 + if "," in line: + name, url = line.split(",", 1) + elif " " in line: + name, url = line.split(" ", 1) + else: + continue + name, url = name.strip(), url.strip() + if name and url.startswith(("http://", "https://", "rtmp://", "rtsp://")): + channels.append(Channel(name=name, url=url)) + return channels + + +def parse(text: str) -> list[Channel]: + """自动识别格式并解析。""" + if "#EXTM3U" in text or "#EXTINF" in text: + return parse_m3u(text) + return parse_txt(text) diff --git a/avtv/sources.py b/avtv/sources.py new file mode 100644 index 0000000..89128e6 --- /dev/null +++ b/avtv/sources.py @@ -0,0 +1,25 @@ +"""直播源地址列表。可自由增删。""" + +# 公开的 m3u / txt 直播源地址 +SOURCES = [ + # iptv-org 中文频道 + "https://iptv-org.github.io/iptv/languages/zho.m3u", + "https://iptv-org.github.io/iptv/countries/cn.m3u", + "https://iptv-org.github.io/iptv/countries/hk.m3u", + "https://iptv-org.github.io/iptv/countries/tw.m3u", + # YanG-1989 维护的国内源 + "https://raw.githubusercontent.com/YanG-1989/m3u/main/Gather.m3u", + "https://raw.githubusercontent.com/YanG-1989/m3u/main/iptv.m3u", + # 其他常见公开源 + "https://raw.githubusercontent.com/iptv-org/iptv/master/streams/cn.m3u", +] + +# 请求超时(秒) +REQUEST_TIMEOUT = 15 + +# User-Agent +USER_AGENT = ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/122.0.0.0 Safari/537.36" +) diff --git a/main.py b/main.py new file mode 100644 index 0000000..7d41e1b --- /dev/null +++ b/main.py @@ -0,0 +1,51 @@ +"""AVTV 主入口:抓取 -> 解析 -> 过滤去重 -> 生成 m3u。""" + +import logging +from pathlib import Path + +from avtv.fetcher import fetch_all +from avtv.filter import process +from avtv.generator import write_m3u +from avtv.parser import Channel, parse +from avtv.sources import SOURCES + +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s [%(levelname)s] %(name)s: %(message)s", +) +log = logging.getLogger("avtv") + +OUTPUT = Path("output/avtv.m3u") + + +def collect() -> list[Channel]: + """抓取并解析所有源。""" + raw_texts = fetch_all(SOURCES) + log.info("successfully fetched %d/%d sources", len(raw_texts), len(SOURCES)) + + all_channels: list[Channel] = [] + for text in raw_texts: + try: + channels = parse(text) + all_channels.extend(channels) + except Exception as exc: # noqa: BLE001 + log.warning("parse failed: %s", exc) + log.info("parsed %d raw channels", len(all_channels)) + return all_channels + + +def run() -> None: + raw = collect() + cleaned = process(raw) + log.info("after filter/dedupe: %d channels", len(cleaned)) + + if not cleaned: + log.error("no valid channels found, skip writing") + return + + out = write_m3u(cleaned, OUTPUT) + log.info("written -> %s", out) + + +if __name__ == "__main__": + run() diff --git a/output/.gitkeep b/output/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..0eb8cae --- /dev/null +++ b/requirements.txt @@ -0,0 +1 @@ +requests>=2.31.0