AI 生成: 一个基于 AI 的全自动短视频生成工具,输入主题即可自动生成脚本、配音、字幕和合成视频。使用 Python + Flask 提供 Web 界面,支持命令行模式。

This commit is contained in:
eeuse
2026-09-28 22:31:39 +08:00
commit 751c99a865
12 changed files with 570 additions and 0 deletions
+79
View File
@@ -0,0 +1,79 @@
# AI 全自动短视频生成器
输入一个主题,自动生成短视频(脚本 → 配音 → 字幕 → 合成)。
## 功能特性
- 🎬 自动生成视频脚本(基于模板 / 可接入大模型)
- 🔊 自动生成配音(pyttsx3 离线 TTS,或接入云服务)
- 📝 自动生成字幕(SRT)
- 🎞️ 自动合成视频(MoviePy)
- 🌐 提供 Web 界面(Flask)
- 💻 提供命令行模式
## 目录结构
```
ai-short-video-generator/
├── app.py # Flask Web 入口
├── cli.py # 命令行入口
├── requirements.txt
├── README.md
├── core/
│ ├── __init__.py
│ ├── script_gen.py # 脚本生成
│ ├── tts.py # 语音合成
│ ├── subtitle.py # 字幕生成
│ └── video.py # 视频合成
├── templates/
│ └── index.html
├── static/
│ ├── style.css
│ └── main.js
└── output/ # 生成结果目录(自动创建)
```
## 安装
```bash
pip install -r requirements.txt
```
> 注意:MoviePy 依赖 ffmpeg,请先安装 ffmpeg 并加入 PATH。
## 使用方式
### 1. Web 模式
```bash
python app.py
```
浏览器打开 http://127.0.0.1:5000 ,输入主题后点击生成。
### 2. 命令行模式
```bash
python cli.py --topic "人工智能的未来" --scenes 5
```
生成的文件位于 `output/` 目录。
## 参数说明
| 参数 | 说明 | 默认值 |
|------|------|--------|
| topic | 视频主题 | 必填 |
| scenes | 场景数量 | 5 |
| voice | 配音语速 | 160 |
| resolution | 视频分辨率 | 1280x720 |
## 扩展
- 接入 OpenAI / 通义千问等大模型:修改 `core/script_gen.py` 中的 `generate_script`。
- 接入云 TTS(如 Azure、讯飞):修改 `core/tts.py`。
- 更换背景视频/图片:修改 `core/video.py`。
## License
MIT
+93
View File
@@ -0,0 +1,93 @@
# -*- coding: utf-8 -*-
"""Flask Web 入口
提供简单的 Web 界面,输入主题后自动生成短视频。
"""
import os
import uuid
import traceback
from flask import Flask, render_template, request, jsonify, send_from_directory
from core.script_gen import generate_script
from core.tts import text_to_speech
from core.subtitle import build_srt
from core.video import compose_video
app = Flask(__name__)
OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "output")
os.makedirs(OUTPUT_DIR, exist_ok=True)
def _probe_duration(path: str) -> float:
"""获取音频时长(秒)"""
from moviepy.editor import AudioFileClip
clip = AudioFileClip(path)
dur = clip.duration
clip.close()
return float(dur)
def run_pipeline(topic: str, scenes: int = 5, rate: int = 160) -> dict:
"""完整生成流水线"""
job_id = uuid.uuid4().hex[:8]
job_dir = os.path.join(OUTPUT_DIR, job_id)
audio_dir = os.path.join(job_dir, "audio")
os.makedirs(audio_dir, exist_ok=True)
script = generate_script(topic, scenes)
audio_paths = []
durations = []
for s in script:
ap = os.path.join(audio_dir, f"{s['index']}.mp3")
text_to_speech(s["text"], ap, rate=rate)
audio_paths.append(ap)
durations.append(_probe_duration(ap))
srt_path = os.path.join(job_dir, "subtitle.srt")
build_srt(script, durations, srt_path)
video_path = os.path.join(job_dir, "video.mp4")
compose_video(script, audio_paths, durations, video_path)
return {
"job_id": job_id,
"video": f"output/{job_id}/video.mp4",
"subtitle": f"output/{job_id}/subtitle.srt",
"script": script,
}
@app.route("/")
def index():
return render_template("index.html")
@app.route("/api/generate", methods=["POST"])
def api_generate():
data = request.get_json(force=True, silent=True) or {}
topic = (data.get("topic") or "").strip()
scenes = int(data.get("scenes") or 5)
rate = int(data.get("rate") or 160)
if not topic:
return jsonify({"ok": False, "error": "请填写主题"}), 400
if scenes < 1 or scenes > 8:
return jsonify({"ok": False, "error": "场景数需在 1-8 之间"}), 400
try:
result = run_pipeline(topic, scenes, rate)
return jsonify({"ok": True, **result})
except Exception as e:
traceback.print_exc()
return jsonify({"ok": False, "error": str(e)}), 500
@app.route("/output/<path:filename>")
def serve_output(filename):
return send_from_directory(OUTPUT_DIR, filename)
if __name__ == "__main__":
app.run(host="0.0.0.0", port=5000, debug=True)
+71
View File
@@ -0,0 +1,71 @@
# -*- coding: utf-8 -*-
"""命令行入口
用法:
python cli.py --topic "人工智能的未来" --scenes 5 --rate 160
"""
import argparse
import os
import sys
import uuid
from core.script_gen import generate_script
from core.tts import text_to_speech
from core.subtitle import build_srt
from core.video import compose_video
OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "output")
def _probe_duration(path: str) -> float:
from moviepy.editor import AudioFileClip
clip = AudioFileClip(path)
dur = clip.duration
clip.close()
return float(dur)
def main():
parser = argparse.ArgumentParser(description="AI 全自动短视频生成器")
parser.add_argument("--topic", required=True, help="视频主题")
parser.add_argument("--scenes", type=int, default=5, help="场景数量(1-8)")
parser.add_argument("--rate", type=int, default=160, help="配音语速")
args = parser.parse_args()
if args.scenes < 1 or args.scenes > 8:
print("场景数需在 1-8 之间")
sys.exit(1)
job_id = uuid.uuid4().hex[:8]
job_dir = os.path.join(OUTPUT_DIR, job_id)
audio_dir = os.path.join(job_dir, "audio")
os.makedirs(audio_dir, exist_ok=True)
print(f"[1/4] 生成脚本...")
script = generate_script(args.topic, args.scenes)
for s in script:
print(f" {s['index']}. {s['text']}")
print(f"[2/4] 合成语音...")
audio_paths, durations = [], []
for s in script:
ap = os.path.join(audio_dir, f"{s['index']}.mp3")
text_to_speech(s["text"], ap, rate=args.rate)
audio_paths.append(ap)
durations.append(_probe_duration(ap))
print(f" {ap} ({durations[-1]:.2f}s)")
print(f"[3/4] 生成字幕...")
srt_path = os.path.join(job_dir, "subtitle.srt")
build_srt(script, durations, srt_path)
print(f" {srt_path}")
print(f"[4/4] 合成视频...")
video_path = os.path.join(job_dir, "video.mp4")
compose_video(script, audio_paths, durations, video_path)
print(f"\n✅ 完成:{video_path}")
if __name__ == "__main__":
main()
+1
View File
@@ -0,0 +1 @@
# core package
+42
View File
@@ -0,0 +1,42 @@
# -*- coding: utf-8 -*-
"""脚本生成模块
默认使用模板方式生成脚本。
如需接入大模型(OpenAI、通义千问等),只需替换 generate_script 的实现。
"""
def generate_script(topic: str, scenes: int = 5) -> list:
"""根据主题生成分场景脚本。
Args:
topic: 视频主题
scenes: 场景数量
Returns:
list[dict]: 每个元素形如 {"index": 1, "text": "..."}
"""
templates = [
"大家好,今天我们来聊一聊{topic}。",
"首先,{topic}正在改变我们的生活方式。",
"其次,{topic}背后蕴含着巨大的机会与挑战。",
"再次,掌握{topic}的关键在于持续学习与实践。",
"最后,让我们一起拥抱{topic}带来的未来。",
"关于{topic},还有很多值得探索的方向。",
"相信不久的将来,{topic}会走进千家万户。",
"感谢观看,我们下期再见!",
]
result = []
for i in range(scenes):
tpl = templates[i % len(templates)]
result.append({
"index": i + 1,
"text": tpl.format(topic=topic),
})
return result
if __name__ == "__main__":
for s in generate_script("人工智能", 3):
print(s)
+48
View File
@@ -0,0 +1,48 @@
# -*- coding: utf-8 -*-
"""字幕生成模块
根据每段文本的时长估算字幕时间轴,生成 SRT 文件。
"""
import os
def _fmt_time(seconds: float) -> str:
"""将秒数转换为 SRT 时间格式 HH:MM:SS,mmm"""
ms = int((seconds - int(seconds)) * 1000)
s = int(seconds) % 60
m = int(seconds) // 60 % 60
h = int(seconds) // 3600
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
def build_srt(scenes: list, durations: list, out_path: str) -> str:
"""根据场景文本和每段时长生成 SRT 字幕。
Args:
scenes: [{"index":1,"text":"..."}, ...]
durations: 每段音频时长(秒),与 scenes 一一对应
out_path: 输出 .srt 路径
Returns:
str: 输出文件路径
"""
os.makedirs(os.path.dirname(out_path), exist_ok=True)
lines = []
cursor = 0.0
for i, (scene, dur) in enumerate(zip(scenes, durations), start=1):
start = cursor
end = cursor + dur
lines.append(str(i))
lines.append(f"{_fmt_time(start)} --> {_fmt_time(end)}")
lines.append(scene["text"])
lines.append("")
cursor = end
with open(out_path, "w", encoding="utf-8") as f:
f.write("\n".join(lines))
return out_path
if __name__ == "__main__":
scenes = [{"index": 1, "text": "你好世界"}, {"index": 2, "text": "再见世界"}]
print(build_srt(scenes, [2.0, 3.0], "output/subtitle/demo.srt"))
+33
View File
@@ -0,0 +1,33 @@
# -*- coding: utf-8 -*-
"""语音合成模块(pyttsx3 离线 TTS)"""
import os
import pyttsx3
def text_to_speech(text: str, out_path: str, rate: int = 160) -> str:
"""将文本合成为音频文件。
Args:
text: 待合成文本
out_path: 输出音频路径(如 output/audio/1.mp3)
rate: 语速
Returns:
str: 输出文件路径
"""
os.makedirs(os.path.dirname(out_path), exist_ok=True)
engine = pyttsx3.init()
engine.setProperty("rate", rate)
# 选择第一个可用语音(不同系统可能不同)
voices = engine.getProperty("voices")
if voices:
engine.setProperty("voice", voices[0].id)
engine.save_to_file(text, out_path)
engine.runAndWait()
engine.stop()
return out_path
if __name__ == "__main__":
print(text_to_speech("你好,这是一段测试语音。", "output/audio/test.mp3"))
+67
View File
@@ -0,0 +1,67 @@
# -*- coding: utf-8 -*-
"""视频合成模块
使用 MoviePy 将图片背景 + 音频 + 字幕合成为最终视频。
"""
import os
from moviepy.editor import (
ImageClip,
AudioFileClip,
CompositeVideoClip,
concatenate_videoclips,
ColorClip,
)
from moviepy.video.tools.subtitles import SubtitlesClip
from moviepy.editor import TextClip
def _make_bg(size, duration, color=(20, 20, 40)):
"""生成纯色背景片段"""
return ColorClip(size=size, color=color, duration=duration)
def _make_subtitle_clip(text: str, size, duration: float):
"""生成一段字幕文本片段"""
w, h = size
try:
txt = TextClip(text, fontsize=48, color="white", font="Arial",
size=(w - 100, None), method="caption")
except Exception:
# 若系统找不到字体,退化为默认字体
txt = TextClip(text, fontsize=48, color="white",
size=(w - 100, None), method="caption")
txt = txt.set_position(("center", h - 180)).set_duration(duration)
return txt
def compose_video(scenes: list, audio_paths: list, durations: list,
out_path: str, resolution=(1280, 720)) -> str:
"""合成最终视频。
Args:
scenes: 场景列表
audio_paths: 每个场景对应的音频路径
durations: 每个场景的时长
out_path: 输出 mp4 路径
resolution: (width, height)
Returns:
str: 输出文件路径
"""
os.makedirs(os.path.dirname(out_path), exist_ok=True)
clips = []
for scene, audio, dur in zip(scenes, audio_paths, durations):
bg = _make_bg(resolution, dur)
sub = _make_subtitle_clip(scene["text"], resolution, dur)
audio_clip = AudioFileClip(audio)
clip = CompositeVideoClip([bg, sub]).set_audio(audio_clip)
clips.append(clip)
final = concatenate_videoclips(clips, method="compose")
final.write_videofile(out_path, fps=24, codec="libx264", audio_codec="aac")
return out_path
if __name__ == "__main__":
print("请通过 cli.py 或 app.py 调用")
+5
View File
@@ -0,0 +1,5 @@
Flask==3.0.0
moviepy==1.0.3
pyttsx3==2.90
Pillow==10.2.0
numpy==1.26.4
+46
View File
@@ -0,0 +1,46 @@
const btn = document.getElementById('btn');
const statusEl = document.getElementById('status');
const resultEl = document.getElementById('result');
const videoEl = document.getElementById('video');
const downloadEl = document.getElementById('download');
const scriptEl = document.getElementById('script');
btn.addEventListener('click', async () => {
const topic = document.getElementById('topic').value.trim();
const scenes = parseInt(document.getElementById('scenes').value, 10) || 5;
const rate = parseInt(document.getElementById('rate').value, 10) || 160;
if (!topic) {
statusEl.textContent = '请填写主题';
return;
}
btn.disabled = true;
resultEl.classList.add('hidden');
statusEl.textContent = '生成中,请稍候(可能需要 1-3 分钟)...';
try {
const res = await fetch('/api/generate', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ topic, scenes, rate }),
});
const data = await res.json();
if (!data.ok) throw new Error(data.error || '生成失败');
statusEl.textContent = '完成!';
videoEl.src = '/' + data.video;
downloadEl.href = '/' + data.video;
scriptEl.innerHTML = '';
data.script.forEach((s) => {
const li = document.createElement('li');
li.textContent = s.text;
scriptEl.appendChild(li);
});
resultEl.classList.remove('hidden');
} catch (e) {
statusEl.textContent = '错误:' + e.message;
} finally {
btn.disabled = false;
}
});
+45
View File
@@ -0,0 +1,45 @@
* { box-sizing: border-box; }
body {
margin: 0;
font-family: -apple-system, "Segoe UI", "PingFang SC", "Microsoft YaHei", sans-serif;
background: linear-gradient(135deg, #1e1e2f 0%, #2a2a4a 100%);
color: #eaeaea;
min-height: 100vh;
}
.container {
max-width: 720px;
margin: 40px auto;
padding: 32px;
background: rgba(255, 255, 255, 0.05);
border-radius: 16px;
box-shadow: 0 8px 32px rgba(0, 0, 0, 0.4);
}
h1 { margin-top: 0; font-size: 24px; }
.sub { color: #aaa; margin-bottom: 24px; }
.form { display: flex; flex-direction: column; gap: 8px; }
.form label { font-size: 14px; color: #bbb; }
.form input {
padding: 10px 12px;
border-radius: 8px;
border: 1px solid #444;
background: #1a1a2a;
color: #eee;
font-size: 14px;
}
.form button {
margin-top: 12px;
padding: 12px;
border: none;
border-radius: 8px;
background: linear-gradient(90deg, #6a5acd, #8a2be2);
color: #fff;
font-size: 16px;
cursor: pointer;
transition: opacity 0.2s;
}
.form button:disabled { opacity: 0.5; cursor: not-allowed; }
.status { margin-top: 20px; color: #8fd3ff; min-height: 20px; }
.result { margin-top: 24px; }
.result video { width: 100%; border-radius: 12px; background: #000; }
.result a { color: #8fd3ff; }
.hidden { display: none; }
+40
View File
@@ -0,0 +1,40 @@
<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<title>AI 全自动短视频生成器</title>
<link rel="stylesheet" href="/static/style.css" />
</head>
<body>
<div class="container">
<h1>🎬 AI 全自动短视频生成器</h1>
<p class="sub">输入一个主题,自动生成脚本、配音、字幕和视频</p>
<div class="form">
<label>主题</label>
<input id="topic" type="text" placeholder="例如:人工智能的未来" />
<label>场景数量(1-8)</label>
<input id="scenes" type="number" value="5" min="1" max="8" />
<label>语速</label>
<input id="rate" type="number" value="160" min="80" max="260" />
<button id="btn">开始生成</button>
</div>
<div id="status" class="status"></div>
<div id="result" class="result hidden">
<h2>生成结果</h2>
<video id="video" controls></video>
<p><a id="download" href="#" download>下载视频</a></p>
<h3>脚本</h3>
<ol id="script"></ol>
</div>
</div>
<script src="/static/main.js"></script>
</body>
</html>