AI 生成: 一个基于 AI 的全自动短视频生成工具,输入主题即可自动生成脚本、配音、字幕和合成视频。使用 Python + Flask 提供 Web 界面,支持命令行模式。
This commit is contained in:
@@ -0,0 +1,79 @@
|
||||
# AI 全自动短视频生成器
|
||||
|
||||
输入一个主题,自动生成短视频(脚本 → 配音 → 字幕 → 合成)。
|
||||
|
||||
## 功能特性
|
||||
|
||||
- 🎬 自动生成视频脚本(基于模板 / 可接入大模型)
|
||||
- 🔊 自动生成配音(pyttsx3 离线 TTS,或接入云服务)
|
||||
- 📝 自动生成字幕(SRT)
|
||||
- 🎞️ 自动合成视频(MoviePy)
|
||||
- 🌐 提供 Web 界面(Flask)
|
||||
- 💻 提供命令行模式
|
||||
|
||||
## 目录结构
|
||||
|
||||
```
|
||||
ai-short-video-generator/
|
||||
├── app.py # Flask Web 入口
|
||||
├── cli.py # 命令行入口
|
||||
├── requirements.txt
|
||||
├── README.md
|
||||
├── core/
|
||||
│ ├── __init__.py
|
||||
│ ├── script_gen.py # 脚本生成
|
||||
│ ├── tts.py # 语音合成
|
||||
│ ├── subtitle.py # 字幕生成
|
||||
│ └── video.py # 视频合成
|
||||
├── templates/
|
||||
│ └── index.html
|
||||
├── static/
|
||||
│ ├── style.css
|
||||
│ └── main.js
|
||||
└── output/ # 生成结果目录(自动创建)
|
||||
```
|
||||
|
||||
## 安装
|
||||
|
||||
```bash
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
|
||||
> 注意:MoviePy 依赖 ffmpeg,请先安装 ffmpeg 并加入 PATH。
|
||||
|
||||
## 使用方式
|
||||
|
||||
### 1. Web 模式
|
||||
|
||||
```bash
|
||||
python app.py
|
||||
```
|
||||
|
||||
浏览器打开 http://127.0.0.1:5000 ,输入主题后点击生成。
|
||||
|
||||
### 2. 命令行模式
|
||||
|
||||
```bash
|
||||
python cli.py --topic "人工智能的未来" --scenes 5
|
||||
```
|
||||
|
||||
生成的文件位于 `output/` 目录。
|
||||
|
||||
## 参数说明
|
||||
|
||||
| 参数 | 说明 | 默认值 |
|
||||
|------|------|--------|
|
||||
| topic | 视频主题 | 必填 |
|
||||
| scenes | 场景数量 | 5 |
|
||||
| voice | 配音语速 | 160 |
|
||||
| resolution | 视频分辨率 | 1280x720 |
|
||||
|
||||
## 扩展
|
||||
|
||||
- 接入 OpenAI / 通义千问等大模型:修改 `core/script_gen.py` 中的 `generate_script`。
|
||||
- 接入云 TTS(如 Azure、讯飞):修改 `core/tts.py`。
|
||||
- 更换背景视频/图片:修改 `core/video.py`。
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
@@ -0,0 +1,93 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Flask Web 入口
|
||||
|
||||
提供简单的 Web 界面,输入主题后自动生成短视频。
|
||||
"""
|
||||
|
||||
import os
|
||||
import uuid
|
||||
import traceback
|
||||
from flask import Flask, render_template, request, jsonify, send_from_directory
|
||||
|
||||
from core.script_gen import generate_script
|
||||
from core.tts import text_to_speech
|
||||
from core.subtitle import build_srt
|
||||
from core.video import compose_video
|
||||
|
||||
app = Flask(__name__)
|
||||
OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "output")
|
||||
os.makedirs(OUTPUT_DIR, exist_ok=True)
|
||||
|
||||
|
||||
def _probe_duration(path: str) -> float:
|
||||
"""获取音频时长(秒)"""
|
||||
from moviepy.editor import AudioFileClip
|
||||
clip = AudioFileClip(path)
|
||||
dur = clip.duration
|
||||
clip.close()
|
||||
return float(dur)
|
||||
|
||||
|
||||
def run_pipeline(topic: str, scenes: int = 5, rate: int = 160) -> dict:
|
||||
"""完整生成流水线"""
|
||||
job_id = uuid.uuid4().hex[:8]
|
||||
job_dir = os.path.join(OUTPUT_DIR, job_id)
|
||||
audio_dir = os.path.join(job_dir, "audio")
|
||||
os.makedirs(audio_dir, exist_ok=True)
|
||||
|
||||
script = generate_script(topic, scenes)
|
||||
|
||||
audio_paths = []
|
||||
durations = []
|
||||
for s in script:
|
||||
ap = os.path.join(audio_dir, f"{s['index']}.mp3")
|
||||
text_to_speech(s["text"], ap, rate=rate)
|
||||
audio_paths.append(ap)
|
||||
durations.append(_probe_duration(ap))
|
||||
|
||||
srt_path = os.path.join(job_dir, "subtitle.srt")
|
||||
build_srt(script, durations, srt_path)
|
||||
|
||||
video_path = os.path.join(job_dir, "video.mp4")
|
||||
compose_video(script, audio_paths, durations, video_path)
|
||||
|
||||
return {
|
||||
"job_id": job_id,
|
||||
"video": f"output/{job_id}/video.mp4",
|
||||
"subtitle": f"output/{job_id}/subtitle.srt",
|
||||
"script": script,
|
||||
}
|
||||
|
||||
|
||||
@app.route("/")
|
||||
def index():
|
||||
return render_template("index.html")
|
||||
|
||||
|
||||
@app.route("/api/generate", methods=["POST"])
|
||||
def api_generate():
|
||||
data = request.get_json(force=True, silent=True) or {}
|
||||
topic = (data.get("topic") or "").strip()
|
||||
scenes = int(data.get("scenes") or 5)
|
||||
rate = int(data.get("rate") or 160)
|
||||
|
||||
if not topic:
|
||||
return jsonify({"ok": False, "error": "请填写主题"}), 400
|
||||
if scenes < 1 or scenes > 8:
|
||||
return jsonify({"ok": False, "error": "场景数需在 1-8 之间"}), 400
|
||||
|
||||
try:
|
||||
result = run_pipeline(topic, scenes, rate)
|
||||
return jsonify({"ok": True, **result})
|
||||
except Exception as e:
|
||||
traceback.print_exc()
|
||||
return jsonify({"ok": False, "error": str(e)}), 500
|
||||
|
||||
|
||||
@app.route("/output/<path:filename>")
|
||||
def serve_output(filename):
|
||||
return send_from_directory(OUTPUT_DIR, filename)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app.run(host="0.0.0.0", port=5000, debug=True)
|
||||
@@ -0,0 +1,71 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""命令行入口
|
||||
|
||||
用法:
|
||||
python cli.py --topic "人工智能的未来" --scenes 5 --rate 160
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import sys
|
||||
import uuid
|
||||
|
||||
from core.script_gen import generate_script
|
||||
from core.tts import text_to_speech
|
||||
from core.subtitle import build_srt
|
||||
from core.video import compose_video
|
||||
|
||||
OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "output")
|
||||
|
||||
|
||||
def _probe_duration(path: str) -> float:
|
||||
from moviepy.editor import AudioFileClip
|
||||
clip = AudioFileClip(path)
|
||||
dur = clip.duration
|
||||
clip.close()
|
||||
return float(dur)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="AI 全自动短视频生成器")
|
||||
parser.add_argument("--topic", required=True, help="视频主题")
|
||||
parser.add_argument("--scenes", type=int, default=5, help="场景数量(1-8)")
|
||||
parser.add_argument("--rate", type=int, default=160, help="配音语速")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.scenes < 1 or args.scenes > 8:
|
||||
print("场景数需在 1-8 之间")
|
||||
sys.exit(1)
|
||||
|
||||
job_id = uuid.uuid4().hex[:8]
|
||||
job_dir = os.path.join(OUTPUT_DIR, job_id)
|
||||
audio_dir = os.path.join(job_dir, "audio")
|
||||
os.makedirs(audio_dir, exist_ok=True)
|
||||
|
||||
print(f"[1/4] 生成脚本...")
|
||||
script = generate_script(args.topic, args.scenes)
|
||||
for s in script:
|
||||
print(f" {s['index']}. {s['text']}")
|
||||
|
||||
print(f"[2/4] 合成语音...")
|
||||
audio_paths, durations = [], []
|
||||
for s in script:
|
||||
ap = os.path.join(audio_dir, f"{s['index']}.mp3")
|
||||
text_to_speech(s["text"], ap, rate=args.rate)
|
||||
audio_paths.append(ap)
|
||||
durations.append(_probe_duration(ap))
|
||||
print(f" {ap} ({durations[-1]:.2f}s)")
|
||||
|
||||
print(f"[3/4] 生成字幕...")
|
||||
srt_path = os.path.join(job_dir, "subtitle.srt")
|
||||
build_srt(script, durations, srt_path)
|
||||
print(f" {srt_path}")
|
||||
|
||||
print(f"[4/4] 合成视频...")
|
||||
video_path = os.path.join(job_dir, "video.mp4")
|
||||
compose_video(script, audio_paths, durations, video_path)
|
||||
print(f"\n✅ 完成:{video_path}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1 @@
|
||||
# core package
|
||||
@@ -0,0 +1,42 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""脚本生成模块
|
||||
|
||||
默认使用模板方式生成脚本。
|
||||
如需接入大模型(OpenAI、通义千问等),只需替换 generate_script 的实现。
|
||||
"""
|
||||
|
||||
|
||||
def generate_script(topic: str, scenes: int = 5) -> list:
|
||||
"""根据主题生成分场景脚本。
|
||||
|
||||
Args:
|
||||
topic: 视频主题
|
||||
scenes: 场景数量
|
||||
|
||||
Returns:
|
||||
list[dict]: 每个元素形如 {"index": 1, "text": "..."}
|
||||
"""
|
||||
templates = [
|
||||
"大家好,今天我们来聊一聊{topic}。",
|
||||
"首先,{topic}正在改变我们的生活方式。",
|
||||
"其次,{topic}背后蕴含着巨大的机会与挑战。",
|
||||
"再次,掌握{topic}的关键在于持续学习与实践。",
|
||||
"最后,让我们一起拥抱{topic}带来的未来。",
|
||||
"关于{topic},还有很多值得探索的方向。",
|
||||
"相信不久的将来,{topic}会走进千家万户。",
|
||||
"感谢观看,我们下期再见!",
|
||||
]
|
||||
|
||||
result = []
|
||||
for i in range(scenes):
|
||||
tpl = templates[i % len(templates)]
|
||||
result.append({
|
||||
"index": i + 1,
|
||||
"text": tpl.format(topic=topic),
|
||||
})
|
||||
return result
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
for s in generate_script("人工智能", 3):
|
||||
print(s)
|
||||
@@ -0,0 +1,48 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""字幕生成模块
|
||||
|
||||
根据每段文本的时长估算字幕时间轴,生成 SRT 文件。
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
|
||||
def _fmt_time(seconds: float) -> str:
|
||||
"""将秒数转换为 SRT 时间格式 HH:MM:SS,mmm"""
|
||||
ms = int((seconds - int(seconds)) * 1000)
|
||||
s = int(seconds) % 60
|
||||
m = int(seconds) // 60 % 60
|
||||
h = int(seconds) // 3600
|
||||
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
|
||||
|
||||
|
||||
def build_srt(scenes: list, durations: list, out_path: str) -> str:
|
||||
"""根据场景文本和每段时长生成 SRT 字幕。
|
||||
|
||||
Args:
|
||||
scenes: [{"index":1,"text":"..."}, ...]
|
||||
durations: 每段音频时长(秒),与 scenes 一一对应
|
||||
out_path: 输出 .srt 路径
|
||||
|
||||
Returns:
|
||||
str: 输出文件路径
|
||||
"""
|
||||
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
||||
lines = []
|
||||
cursor = 0.0
|
||||
for i, (scene, dur) in enumerate(zip(scenes, durations), start=1):
|
||||
start = cursor
|
||||
end = cursor + dur
|
||||
lines.append(str(i))
|
||||
lines.append(f"{_fmt_time(start)} --> {_fmt_time(end)}")
|
||||
lines.append(scene["text"])
|
||||
lines.append("")
|
||||
cursor = end
|
||||
with open(out_path, "w", encoding="utf-8") as f:
|
||||
f.write("\n".join(lines))
|
||||
return out_path
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
scenes = [{"index": 1, "text": "你好世界"}, {"index": 2, "text": "再见世界"}]
|
||||
print(build_srt(scenes, [2.0, 3.0], "output/subtitle/demo.srt"))
|
||||
+33
@@ -0,0 +1,33 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""语音合成模块(pyttsx3 离线 TTS)"""
|
||||
|
||||
import os
|
||||
import pyttsx3
|
||||
|
||||
|
||||
def text_to_speech(text: str, out_path: str, rate: int = 160) -> str:
|
||||
"""将文本合成为音频文件。
|
||||
|
||||
Args:
|
||||
text: 待合成文本
|
||||
out_path: 输出音频路径(如 output/audio/1.mp3)
|
||||
rate: 语速
|
||||
|
||||
Returns:
|
||||
str: 输出文件路径
|
||||
"""
|
||||
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
||||
engine = pyttsx3.init()
|
||||
engine.setProperty("rate", rate)
|
||||
# 选择第一个可用语音(不同系统可能不同)
|
||||
voices = engine.getProperty("voices")
|
||||
if voices:
|
||||
engine.setProperty("voice", voices[0].id)
|
||||
engine.save_to_file(text, out_path)
|
||||
engine.runAndWait()
|
||||
engine.stop()
|
||||
return out_path
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print(text_to_speech("你好,这是一段测试语音。", "output/audio/test.mp3"))
|
||||
@@ -0,0 +1,67 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
"""视频合成模块
|
||||
|
||||
使用 MoviePy 将图片背景 + 音频 + 字幕合成为最终视频。
|
||||
"""
|
||||
|
||||
import os
|
||||
from moviepy.editor import (
|
||||
ImageClip,
|
||||
AudioFileClip,
|
||||
CompositeVideoClip,
|
||||
concatenate_videoclips,
|
||||
ColorClip,
|
||||
)
|
||||
from moviepy.video.tools.subtitles import SubtitlesClip
|
||||
from moviepy.editor import TextClip
|
||||
|
||||
|
||||
def _make_bg(size, duration, color=(20, 20, 40)):
|
||||
"""生成纯色背景片段"""
|
||||
return ColorClip(size=size, color=color, duration=duration)
|
||||
|
||||
|
||||
def _make_subtitle_clip(text: str, size, duration: float):
|
||||
"""生成一段字幕文本片段"""
|
||||
w, h = size
|
||||
try:
|
||||
txt = TextClip(text, fontsize=48, color="white", font="Arial",
|
||||
size=(w - 100, None), method="caption")
|
||||
except Exception:
|
||||
# 若系统找不到字体,退化为默认字体
|
||||
txt = TextClip(text, fontsize=48, color="white",
|
||||
size=(w - 100, None), method="caption")
|
||||
txt = txt.set_position(("center", h - 180)).set_duration(duration)
|
||||
return txt
|
||||
|
||||
|
||||
def compose_video(scenes: list, audio_paths: list, durations: list,
|
||||
out_path: str, resolution=(1280, 720)) -> str:
|
||||
"""合成最终视频。
|
||||
|
||||
Args:
|
||||
scenes: 场景列表
|
||||
audio_paths: 每个场景对应的音频路径
|
||||
durations: 每个场景的时长
|
||||
out_path: 输出 mp4 路径
|
||||
resolution: (width, height)
|
||||
|
||||
Returns:
|
||||
str: 输出文件路径
|
||||
"""
|
||||
os.makedirs(os.path.dirname(out_path), exist_ok=True)
|
||||
clips = []
|
||||
for scene, audio, dur in zip(scenes, audio_paths, durations):
|
||||
bg = _make_bg(resolution, dur)
|
||||
sub = _make_subtitle_clip(scene["text"], resolution, dur)
|
||||
audio_clip = AudioFileClip(audio)
|
||||
clip = CompositeVideoClip([bg, sub]).set_audio(audio_clip)
|
||||
clips.append(clip)
|
||||
|
||||
final = concatenate_videoclips(clips, method="compose")
|
||||
final.write_videofile(out_path, fps=24, codec="libx264", audio_codec="aac")
|
||||
return out_path
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print("请通过 cli.py 或 app.py 调用")
|
||||
@@ -0,0 +1,5 @@
|
||||
Flask==3.0.0
|
||||
moviepy==1.0.3
|
||||
pyttsx3==2.90
|
||||
Pillow==10.2.0
|
||||
numpy==1.26.4
|
||||
@@ -0,0 +1,46 @@
|
||||
const btn = document.getElementById('btn');
|
||||
const statusEl = document.getElementById('status');
|
||||
const resultEl = document.getElementById('result');
|
||||
const videoEl = document.getElementById('video');
|
||||
const downloadEl = document.getElementById('download');
|
||||
const scriptEl = document.getElementById('script');
|
||||
|
||||
btn.addEventListener('click', async () => {
|
||||
const topic = document.getElementById('topic').value.trim();
|
||||
const scenes = parseInt(document.getElementById('scenes').value, 10) || 5;
|
||||
const rate = parseInt(document.getElementById('rate').value, 10) || 160;
|
||||
|
||||
if (!topic) {
|
||||
statusEl.textContent = '请填写主题';
|
||||
return;
|
||||
}
|
||||
|
||||
btn.disabled = true;
|
||||
resultEl.classList.add('hidden');
|
||||
statusEl.textContent = '生成中,请稍候(可能需要 1-3 分钟)...';
|
||||
|
||||
try {
|
||||
const res = await fetch('/api/generate', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ topic, scenes, rate }),
|
||||
});
|
||||
const data = await res.json();
|
||||
if (!data.ok) throw new Error(data.error || '生成失败');
|
||||
|
||||
statusEl.textContent = '完成!';
|
||||
videoEl.src = '/' + data.video;
|
||||
downloadEl.href = '/' + data.video;
|
||||
scriptEl.innerHTML = '';
|
||||
data.script.forEach((s) => {
|
||||
const li = document.createElement('li');
|
||||
li.textContent = s.text;
|
||||
scriptEl.appendChild(li);
|
||||
});
|
||||
resultEl.classList.remove('hidden');
|
||||
} catch (e) {
|
||||
statusEl.textContent = '错误:' + e.message;
|
||||
} finally {
|
||||
btn.disabled = false;
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1,45 @@
|
||||
* { box-sizing: border-box; }
|
||||
body {
|
||||
margin: 0;
|
||||
font-family: -apple-system, "Segoe UI", "PingFang SC", "Microsoft YaHei", sans-serif;
|
||||
background: linear-gradient(135deg, #1e1e2f 0%, #2a2a4a 100%);
|
||||
color: #eaeaea;
|
||||
min-height: 100vh;
|
||||
}
|
||||
.container {
|
||||
max-width: 720px;
|
||||
margin: 40px auto;
|
||||
padding: 32px;
|
||||
background: rgba(255, 255, 255, 0.05);
|
||||
border-radius: 16px;
|
||||
box-shadow: 0 8px 32px rgba(0, 0, 0, 0.4);
|
||||
}
|
||||
h1 { margin-top: 0; font-size: 24px; }
|
||||
.sub { color: #aaa; margin-bottom: 24px; }
|
||||
.form { display: flex; flex-direction: column; gap: 8px; }
|
||||
.form label { font-size: 14px; color: #bbb; }
|
||||
.form input {
|
||||
padding: 10px 12px;
|
||||
border-radius: 8px;
|
||||
border: 1px solid #444;
|
||||
background: #1a1a2a;
|
||||
color: #eee;
|
||||
font-size: 14px;
|
||||
}
|
||||
.form button {
|
||||
margin-top: 12px;
|
||||
padding: 12px;
|
||||
border: none;
|
||||
border-radius: 8px;
|
||||
background: linear-gradient(90deg, #6a5acd, #8a2be2);
|
||||
color: #fff;
|
||||
font-size: 16px;
|
||||
cursor: pointer;
|
||||
transition: opacity 0.2s;
|
||||
}
|
||||
.form button:disabled { opacity: 0.5; cursor: not-allowed; }
|
||||
.status { margin-top: 20px; color: #8fd3ff; min-height: 20px; }
|
||||
.result { margin-top: 24px; }
|
||||
.result video { width: 100%; border-radius: 12px; background: #000; }
|
||||
.result a { color: #8fd3ff; }
|
||||
.hidden { display: none; }
|
||||
@@ -0,0 +1,40 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="zh-CN">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<title>AI 全自动短视频生成器</title>
|
||||
<link rel="stylesheet" href="/static/style.css" />
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
<h1>🎬 AI 全自动短视频生成器</h1>
|
||||
<p class="sub">输入一个主题,自动生成脚本、配音、字幕和视频</p>
|
||||
|
||||
<div class="form">
|
||||
<label>主题</label>
|
||||
<input id="topic" type="text" placeholder="例如:人工智能的未来" />
|
||||
|
||||
<label>场景数量(1-8)</label>
|
||||
<input id="scenes" type="number" value="5" min="1" max="8" />
|
||||
|
||||
<label>语速</label>
|
||||
<input id="rate" type="number" value="160" min="80" max="260" />
|
||||
|
||||
<button id="btn">开始生成</button>
|
||||
</div>
|
||||
|
||||
<div id="status" class="status"></div>
|
||||
|
||||
<div id="result" class="result hidden">
|
||||
<h2>生成结果</h2>
|
||||
<video id="video" controls></video>
|
||||
<p><a id="download" href="#" download>下载视频</a></p>
|
||||
<h3>脚本</h3>
|
||||
<ol id="script"></ol>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<script src="/static/main.js"></script>
|
||||
</body>
|
||||
</html>
|
||||
Reference in New Issue
Block a user