commit 751c99a865e06eabbd7950a360d6d07cfd89f56f Author: eeuse Date: Mon Sep 28 22:31:39 2026 +0800 AI 生成: 一个基于 AI 的全自动短视频生成工具,输入主题即可自动生成脚本、配音、字幕和合成视频。使用 Python + Flask 提供 Web 界面,支持命令行模式。 diff --git a/README.md b/README.md new file mode 100644 index 0000000..0dbe59c --- /dev/null +++ b/README.md @@ -0,0 +1,79 @@ +# AI 全自动短视频生成器 + +输入一个主题,自动生成短视频(脚本 → 配音 → 字幕 → 合成)。 + +## 功能特性 + +- 🎬 自动生成视频脚本(基于模板 / 可接入大模型) +- 🔊 自动生成配音(pyttsx3 离线 TTS,或接入云服务) +- 📝 自动生成字幕(SRT) +- 🎞️ 自动合成视频(MoviePy) +- 🌐 提供 Web 界面(Flask) +- 💻 提供命令行模式 + +## 目录结构 + +``` +ai-short-video-generator/ +├── app.py # Flask Web 入口 +├── cli.py # 命令行入口 +├── requirements.txt +├── README.md +├── core/ +│ ├── __init__.py +│ ├── script_gen.py # 脚本生成 +│ ├── tts.py # 语音合成 +│ ├── subtitle.py # 字幕生成 +│ └── video.py # 视频合成 +├── templates/ +│ └── index.html +├── static/ +│ ├── style.css +│ └── main.js +└── output/ # 生成结果目录(自动创建) +``` + +## 安装 + +```bash +pip install -r requirements.txt +``` + +> 注意:MoviePy 依赖 ffmpeg,请先安装 ffmpeg 并加入 PATH。 + +## 使用方式 + +### 1. Web 模式 + +```bash +python app.py +``` + +浏览器打开 http://127.0.0.1:5000 ,输入主题后点击生成。 + +### 2. 命令行模式 + +```bash +python cli.py --topic "人工智能的未来" --scenes 5 +``` + +生成的文件位于 `output/` 目录。 + +## 参数说明 + +| 参数 | 说明 | 默认值 | +|------|------|--------| +| topic | 视频主题 | 必填 | +| scenes | 场景数量 | 5 | +| voice | 配音语速 | 160 | +| resolution | 视频分辨率 | 1280x720 | + +## 扩展 + +- 接入 OpenAI / 通义千问等大模型:修改 `core/script_gen.py` 中的 `generate_script`。 +- 接入云 TTS(如 Azure、讯飞):修改 `core/tts.py`。 +- 更换背景视频/图片:修改 `core/video.py`。 + +## License + +MIT diff --git a/app.py b/app.py new file mode 100644 index 0000000..72041a0 --- /dev/null +++ b/app.py @@ -0,0 +1,93 @@ +# -*- coding: utf-8 -*- +"""Flask Web 入口 + +提供简单的 Web 界面,输入主题后自动生成短视频。 +""" + +import os +import uuid +import traceback +from flask import Flask, render_template, request, jsonify, send_from_directory + +from core.script_gen import generate_script +from core.tts import text_to_speech +from core.subtitle import build_srt +from core.video import compose_video + +app = Flask(__name__) +OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "output") +os.makedirs(OUTPUT_DIR, exist_ok=True) + + +def _probe_duration(path: str) -> float: + """获取音频时长(秒)""" + from moviepy.editor import AudioFileClip + clip = AudioFileClip(path) + dur = clip.duration + clip.close() + return float(dur) + + +def run_pipeline(topic: str, scenes: int = 5, rate: int = 160) -> dict: + """完整生成流水线""" + job_id = uuid.uuid4().hex[:8] + job_dir = os.path.join(OUTPUT_DIR, job_id) + audio_dir = os.path.join(job_dir, "audio") + os.makedirs(audio_dir, exist_ok=True) + + script = generate_script(topic, scenes) + + audio_paths = [] + durations = [] + for s in script: + ap = os.path.join(audio_dir, f"{s['index']}.mp3") + text_to_speech(s["text"], ap, rate=rate) + audio_paths.append(ap) + durations.append(_probe_duration(ap)) + + srt_path = os.path.join(job_dir, "subtitle.srt") + build_srt(script, durations, srt_path) + + video_path = os.path.join(job_dir, "video.mp4") + compose_video(script, audio_paths, durations, video_path) + + return { + "job_id": job_id, + "video": f"output/{job_id}/video.mp4", + "subtitle": f"output/{job_id}/subtitle.srt", + "script": script, + } + + +@app.route("/") +def index(): + return render_template("index.html") + + +@app.route("/api/generate", methods=["POST"]) +def api_generate(): + data = request.get_json(force=True, silent=True) or {} + topic = (data.get("topic") or "").strip() + scenes = int(data.get("scenes") or 5) + rate = int(data.get("rate") or 160) + + if not topic: + return jsonify({"ok": False, "error": "请填写主题"}), 400 + if scenes < 1 or scenes > 8: + return jsonify({"ok": False, "error": "场景数需在 1-8 之间"}), 400 + + try: + result = run_pipeline(topic, scenes, rate) + return jsonify({"ok": True, **result}) + except Exception as e: + traceback.print_exc() + return jsonify({"ok": False, "error": str(e)}), 500 + + +@app.route("/output/") +def serve_output(filename): + return send_from_directory(OUTPUT_DIR, filename) + + +if __name__ == "__main__": + app.run(host="0.0.0.0", port=5000, debug=True) diff --git a/cli.py b/cli.py new file mode 100644 index 0000000..e960756 --- /dev/null +++ b/cli.py @@ -0,0 +1,71 @@ +# -*- coding: utf-8 -*- +"""命令行入口 + +用法: + python cli.py --topic "人工智能的未来" --scenes 5 --rate 160 +""" + +import argparse +import os +import sys +import uuid + +from core.script_gen import generate_script +from core.tts import text_to_speech +from core.subtitle import build_srt +from core.video import compose_video + +OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "output") + + +def _probe_duration(path: str) -> float: + from moviepy.editor import AudioFileClip + clip = AudioFileClip(path) + dur = clip.duration + clip.close() + return float(dur) + + +def main(): + parser = argparse.ArgumentParser(description="AI 全自动短视频生成器") + parser.add_argument("--topic", required=True, help="视频主题") + parser.add_argument("--scenes", type=int, default=5, help="场景数量(1-8)") + parser.add_argument("--rate", type=int, default=160, help="配音语速") + args = parser.parse_args() + + if args.scenes < 1 or args.scenes > 8: + print("场景数需在 1-8 之间") + sys.exit(1) + + job_id = uuid.uuid4().hex[:8] + job_dir = os.path.join(OUTPUT_DIR, job_id) + audio_dir = os.path.join(job_dir, "audio") + os.makedirs(audio_dir, exist_ok=True) + + print(f"[1/4] 生成脚本...") + script = generate_script(args.topic, args.scenes) + for s in script: + print(f" {s['index']}. {s['text']}") + + print(f"[2/4] 合成语音...") + audio_paths, durations = [], [] + for s in script: + ap = os.path.join(audio_dir, f"{s['index']}.mp3") + text_to_speech(s["text"], ap, rate=args.rate) + audio_paths.append(ap) + durations.append(_probe_duration(ap)) + print(f" {ap} ({durations[-1]:.2f}s)") + + print(f"[3/4] 生成字幕...") + srt_path = os.path.join(job_dir, "subtitle.srt") + build_srt(script, durations, srt_path) + print(f" {srt_path}") + + print(f"[4/4] 合成视频...") + video_path = os.path.join(job_dir, "video.mp4") + compose_video(script, audio_paths, durations, video_path) + print(f"\n✅ 完成:{video_path}") + + +if __name__ == "__main__": + main() diff --git a/core/__init__.py b/core/__init__.py new file mode 100644 index 0000000..97daee7 --- /dev/null +++ b/core/__init__.py @@ -0,0 +1 @@ +# core package diff --git a/core/script_gen.py b/core/script_gen.py new file mode 100644 index 0000000..dd28dc0 --- /dev/null +++ b/core/script_gen.py @@ -0,0 +1,42 @@ +# -*- coding: utf-8 -*- +"""脚本生成模块 + +默认使用模板方式生成脚本。 +如需接入大模型(OpenAI、通义千问等),只需替换 generate_script 的实现。 +""" + + +def generate_script(topic: str, scenes: int = 5) -> list: + """根据主题生成分场景脚本。 + + Args: + topic: 视频主题 + scenes: 场景数量 + + Returns: + list[dict]: 每个元素形如 {"index": 1, "text": "..."} + """ + templates = [ + "大家好,今天我们来聊一聊{topic}。", + "首先,{topic}正在改变我们的生活方式。", + "其次,{topic}背后蕴含着巨大的机会与挑战。", + "再次,掌握{topic}的关键在于持续学习与实践。", + "最后,让我们一起拥抱{topic}带来的未来。", + "关于{topic},还有很多值得探索的方向。", + "相信不久的将来,{topic}会走进千家万户。", + "感谢观看,我们下期再见!", + ] + + result = [] + for i in range(scenes): + tpl = templates[i % len(templates)] + result.append({ + "index": i + 1, + "text": tpl.format(topic=topic), + }) + return result + + +if __name__ == "__main__": + for s in generate_script("人工智能", 3): + print(s) diff --git a/core/subtitle.py b/core/subtitle.py new file mode 100644 index 0000000..253e0a3 --- /dev/null +++ b/core/subtitle.py @@ -0,0 +1,48 @@ +# -*- coding: utf-8 -*- +"""字幕生成模块 + +根据每段文本的时长估算字幕时间轴,生成 SRT 文件。 +""" + +import os + + +def _fmt_time(seconds: float) -> str: + """将秒数转换为 SRT 时间格式 HH:MM:SS,mmm""" + ms = int((seconds - int(seconds)) * 1000) + s = int(seconds) % 60 + m = int(seconds) // 60 % 60 + h = int(seconds) // 3600 + return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}" + + +def build_srt(scenes: list, durations: list, out_path: str) -> str: + """根据场景文本和每段时长生成 SRT 字幕。 + + Args: + scenes: [{"index":1,"text":"..."}, ...] + durations: 每段音频时长(秒),与 scenes 一一对应 + out_path: 输出 .srt 路径 + + Returns: + str: 输出文件路径 + """ + os.makedirs(os.path.dirname(out_path), exist_ok=True) + lines = [] + cursor = 0.0 + for i, (scene, dur) in enumerate(zip(scenes, durations), start=1): + start = cursor + end = cursor + dur + lines.append(str(i)) + lines.append(f"{_fmt_time(start)} --> {_fmt_time(end)}") + lines.append(scene["text"]) + lines.append("") + cursor = end + with open(out_path, "w", encoding="utf-8") as f: + f.write("\n".join(lines)) + return out_path + + +if __name__ == "__main__": + scenes = [{"index": 1, "text": "你好世界"}, {"index": 2, "text": "再见世界"}] + print(build_srt(scenes, [2.0, 3.0], "output/subtitle/demo.srt")) diff --git a/core/tts.py b/core/tts.py new file mode 100644 index 0000000..1ea713b --- /dev/null +++ b/core/tts.py @@ -0,0 +1,33 @@ +# -*- coding: utf-8 -*- +"""语音合成模块(pyttsx3 离线 TTS)""" + +import os +import pyttsx3 + + +def text_to_speech(text: str, out_path: str, rate: int = 160) -> str: + """将文本合成为音频文件。 + + Args: + text: 待合成文本 + out_path: 输出音频路径(如 output/audio/1.mp3) + rate: 语速 + + Returns: + str: 输出文件路径 + """ + os.makedirs(os.path.dirname(out_path), exist_ok=True) + engine = pyttsx3.init() + engine.setProperty("rate", rate) + # 选择第一个可用语音(不同系统可能不同) + voices = engine.getProperty("voices") + if voices: + engine.setProperty("voice", voices[0].id) + engine.save_to_file(text, out_path) + engine.runAndWait() + engine.stop() + return out_path + + +if __name__ == "__main__": + print(text_to_speech("你好,这是一段测试语音。", "output/audio/test.mp3")) diff --git a/core/video.py b/core/video.py new file mode 100644 index 0000000..502d311 --- /dev/null +++ b/core/video.py @@ -0,0 +1,67 @@ +# -*- coding: utf-8 -*- +"""视频合成模块 + +使用 MoviePy 将图片背景 + 音频 + 字幕合成为最终视频。 +""" + +import os +from moviepy.editor import ( + ImageClip, + AudioFileClip, + CompositeVideoClip, + concatenate_videoclips, + ColorClip, +) +from moviepy.video.tools.subtitles import SubtitlesClip +from moviepy.editor import TextClip + + +def _make_bg(size, duration, color=(20, 20, 40)): + """生成纯色背景片段""" + return ColorClip(size=size, color=color, duration=duration) + + +def _make_subtitle_clip(text: str, size, duration: float): + """生成一段字幕文本片段""" + w, h = size + try: + txt = TextClip(text, fontsize=48, color="white", font="Arial", + size=(w - 100, None), method="caption") + except Exception: + # 若系统找不到字体,退化为默认字体 + txt = TextClip(text, fontsize=48, color="white", + size=(w - 100, None), method="caption") + txt = txt.set_position(("center", h - 180)).set_duration(duration) + return txt + + +def compose_video(scenes: list, audio_paths: list, durations: list, + out_path: str, resolution=(1280, 720)) -> str: + """合成最终视频。 + + Args: + scenes: 场景列表 + audio_paths: 每个场景对应的音频路径 + durations: 每个场景的时长 + out_path: 输出 mp4 路径 + resolution: (width, height) + + Returns: + str: 输出文件路径 + """ + os.makedirs(os.path.dirname(out_path), exist_ok=True) + clips = [] + for scene, audio, dur in zip(scenes, audio_paths, durations): + bg = _make_bg(resolution, dur) + sub = _make_subtitle_clip(scene["text"], resolution, dur) + audio_clip = AudioFileClip(audio) + clip = CompositeVideoClip([bg, sub]).set_audio(audio_clip) + clips.append(clip) + + final = concatenate_videoclips(clips, method="compose") + final.write_videofile(out_path, fps=24, codec="libx264", audio_codec="aac") + return out_path + + +if __name__ == "__main__": + print("请通过 cli.py 或 app.py 调用") diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..afbe38d --- /dev/null +++ b/requirements.txt @@ -0,0 +1,5 @@ +Flask==3.0.0 +moviepy==1.0.3 +pyttsx3==2.90 +Pillow==10.2.0 +numpy==1.26.4 diff --git a/static/main.js b/static/main.js new file mode 100644 index 0000000..d6b1713 --- /dev/null +++ b/static/main.js @@ -0,0 +1,46 @@ +const btn = document.getElementById('btn'); +const statusEl = document.getElementById('status'); +const resultEl = document.getElementById('result'); +const videoEl = document.getElementById('video'); +const downloadEl = document.getElementById('download'); +const scriptEl = document.getElementById('script'); + +btn.addEventListener('click', async () => { + const topic = document.getElementById('topic').value.trim(); + const scenes = parseInt(document.getElementById('scenes').value, 10) || 5; + const rate = parseInt(document.getElementById('rate').value, 10) || 160; + + if (!topic) { + statusEl.textContent = '请填写主题'; + return; + } + + btn.disabled = true; + resultEl.classList.add('hidden'); + statusEl.textContent = '生成中,请稍候(可能需要 1-3 分钟)...'; + + try { + const res = await fetch('/api/generate', { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ topic, scenes, rate }), + }); + const data = await res.json(); + if (!data.ok) throw new Error(data.error || '生成失败'); + + statusEl.textContent = '完成!'; + videoEl.src = '/' + data.video; + downloadEl.href = '/' + data.video; + scriptEl.innerHTML = ''; + data.script.forEach((s) => { + const li = document.createElement('li'); + li.textContent = s.text; + scriptEl.appendChild(li); + }); + resultEl.classList.remove('hidden'); + } catch (e) { + statusEl.textContent = '错误:' + e.message; + } finally { + btn.disabled = false; + } +}); diff --git a/static/style.css b/static/style.css new file mode 100644 index 0000000..6820a29 --- /dev/null +++ b/static/style.css @@ -0,0 +1,45 @@ +* { box-sizing: border-box; } +body { + margin: 0; + font-family: -apple-system, "Segoe UI", "PingFang SC", "Microsoft YaHei", sans-serif; + background: linear-gradient(135deg, #1e1e2f 0%, #2a2a4a 100%); + color: #eaeaea; + min-height: 100vh; +} +.container { + max-width: 720px; + margin: 40px auto; + padding: 32px; + background: rgba(255, 255, 255, 0.05); + border-radius: 16px; + box-shadow: 0 8px 32px rgba(0, 0, 0, 0.4); +} +h1 { margin-top: 0; font-size: 24px; } +.sub { color: #aaa; margin-bottom: 24px; } +.form { display: flex; flex-direction: column; gap: 8px; } +.form label { font-size: 14px; color: #bbb; } +.form input { + padding: 10px 12px; + border-radius: 8px; + border: 1px solid #444; + background: #1a1a2a; + color: #eee; + font-size: 14px; +} +.form button { + margin-top: 12px; + padding: 12px; + border: none; + border-radius: 8px; + background: linear-gradient(90deg, #6a5acd, #8a2be2); + color: #fff; + font-size: 16px; + cursor: pointer; + transition: opacity 0.2s; +} +.form button:disabled { opacity: 0.5; cursor: not-allowed; } +.status { margin-top: 20px; color: #8fd3ff; min-height: 20px; } +.result { margin-top: 24px; } +.result video { width: 100%; border-radius: 12px; background: #000; } +.result a { color: #8fd3ff; } +.hidden { display: none; } diff --git a/templates/index.html b/templates/index.html new file mode 100644 index 0000000..b5d755c --- /dev/null +++ b/templates/index.html @@ -0,0 +1,40 @@ + + + + + + AI 全自动短视频生成器 + + + +
+

🎬 AI 全自动短视频生成器

+

输入一个主题,自动生成脚本、配音、字幕和视频

+ +
+ + + + + + + + + + +
+ +
+ + +
+ + + +