AI 生成: 一个基于 AI 的全自动短视频生成工具,输入主题即可自动生成脚本、配音、字幕和合成视频。使用 Python + Flask 提供 Web 界面,支持命令行模式。

This commit is contained in:
eeuse
2026-09-28 22:31:39 +08:00
commit 751c99a865
12 changed files with 570 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
# core package
+42
View File
@@ -0,0 +1,42 @@
# -*- coding: utf-8 -*-
"""脚本生成模块
默认使用模板方式生成脚本。
如需接入大模型(OpenAI、通义千问等),只需替换 generate_script 的实现。
"""
def generate_script(topic: str, scenes: int = 5) -> list:
"""根据主题生成分场景脚本。
Args:
topic: 视频主题
scenes: 场景数量
Returns:
list[dict]: 每个元素形如 {"index": 1, "text": "..."}
"""
templates = [
"大家好,今天我们来聊一聊{topic}。",
"首先,{topic}正在改变我们的生活方式。",
"其次,{topic}背后蕴含着巨大的机会与挑战。",
"再次,掌握{topic}的关键在于持续学习与实践。",
"最后,让我们一起拥抱{topic}带来的未来。",
"关于{topic},还有很多值得探索的方向。",
"相信不久的将来,{topic}会走进千家万户。",
"感谢观看,我们下期再见!",
]
result = []
for i in range(scenes):
tpl = templates[i % len(templates)]
result.append({
"index": i + 1,
"text": tpl.format(topic=topic),
})
return result
if __name__ == "__main__":
for s in generate_script("人工智能", 3):
print(s)
+48
View File
@@ -0,0 +1,48 @@
# -*- coding: utf-8 -*-
"""字幕生成模块
根据每段文本的时长估算字幕时间轴,生成 SRT 文件。
"""
import os
def _fmt_time(seconds: float) -> str:
"""将秒数转换为 SRT 时间格式 HH:MM:SS,mmm"""
ms = int((seconds - int(seconds)) * 1000)
s = int(seconds) % 60
m = int(seconds) // 60 % 60
h = int(seconds) // 3600
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
def build_srt(scenes: list, durations: list, out_path: str) -> str:
"""根据场景文本和每段时长生成 SRT 字幕。
Args:
scenes: [{"index":1,"text":"..."}, ...]
durations: 每段音频时长(秒),与 scenes 一一对应
out_path: 输出 .srt 路径
Returns:
str: 输出文件路径
"""
os.makedirs(os.path.dirname(out_path), exist_ok=True)
lines = []
cursor = 0.0
for i, (scene, dur) in enumerate(zip(scenes, durations), start=1):
start = cursor
end = cursor + dur
lines.append(str(i))
lines.append(f"{_fmt_time(start)} --> {_fmt_time(end)}")
lines.append(scene["text"])
lines.append("")
cursor = end
with open(out_path, "w", encoding="utf-8") as f:
f.write("\n".join(lines))
return out_path
if __name__ == "__main__":
scenes = [{"index": 1, "text": "你好世界"}, {"index": 2, "text": "再见世界"}]
print(build_srt(scenes, [2.0, 3.0], "output/subtitle/demo.srt"))
+33
View File
@@ -0,0 +1,33 @@
# -*- coding: utf-8 -*-
"""语音合成模块(pyttsx3 离线 TTS)"""
import os
import pyttsx3
def text_to_speech(text: str, out_path: str, rate: int = 160) -> str:
"""将文本合成为音频文件。
Args:
text: 待合成文本
out_path: 输出音频路径(如 output/audio/1.mp3)
rate: 语速
Returns:
str: 输出文件路径
"""
os.makedirs(os.path.dirname(out_path), exist_ok=True)
engine = pyttsx3.init()
engine.setProperty("rate", rate)
# 选择第一个可用语音(不同系统可能不同)
voices = engine.getProperty("voices")
if voices:
engine.setProperty("voice", voices[0].id)
engine.save_to_file(text, out_path)
engine.runAndWait()
engine.stop()
return out_path
if __name__ == "__main__":
print(text_to_speech("你好,这是一段测试语音。", "output/audio/test.mp3"))
+67
View File
@@ -0,0 +1,67 @@
# -*- coding: utf-8 -*-
"""视频合成模块
使用 MoviePy 将图片背景 + 音频 + 字幕合成为最终视频。
"""
import os
from moviepy.editor import (
ImageClip,
AudioFileClip,
CompositeVideoClip,
concatenate_videoclips,
ColorClip,
)
from moviepy.video.tools.subtitles import SubtitlesClip
from moviepy.editor import TextClip
def _make_bg(size, duration, color=(20, 20, 40)):
"""生成纯色背景片段"""
return ColorClip(size=size, color=color, duration=duration)
def _make_subtitle_clip(text: str, size, duration: float):
"""生成一段字幕文本片段"""
w, h = size
try:
txt = TextClip(text, fontsize=48, color="white", font="Arial",
size=(w - 100, None), method="caption")
except Exception:
# 若系统找不到字体,退化为默认字体
txt = TextClip(text, fontsize=48, color="white",
size=(w - 100, None), method="caption")
txt = txt.set_position(("center", h - 180)).set_duration(duration)
return txt
def compose_video(scenes: list, audio_paths: list, durations: list,
out_path: str, resolution=(1280, 720)) -> str:
"""合成最终视频。
Args:
scenes: 场景列表
audio_paths: 每个场景对应的音频路径
durations: 每个场景的时长
out_path: 输出 mp4 路径
resolution: (width, height)
Returns:
str: 输出文件路径
"""
os.makedirs(os.path.dirname(out_path), exist_ok=True)
clips = []
for scene, audio, dur in zip(scenes, audio_paths, durations):
bg = _make_bg(resolution, dur)
sub = _make_subtitle_clip(scene["text"], resolution, dur)
audio_clip = AudioFileClip(audio)
clip = CompositeVideoClip([bg, sub]).set_audio(audio_clip)
clips.append(clip)
final = concatenate_videoclips(clips, method="compose")
final.write_videofile(out_path, fps=24, codec="libx264", audio_codec="aac")
return out_path
if __name__ == "__main__":
print("请通过 cli.py 或 app.py 调用")