Popiai-skill仓库
You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
 
 
 
 

219 lines
6.8 KiB

#!/usr/bin/env python3
"""
字幕渲染脚本 - 用 Python PIL 生成透明背景字幕 PNG
解决本地 ffmpeg 未编译 libfreetype/libfontconfig 导致 drawtext 不可用的问题
使用方式:
python render_subtitles.py --script voiceover_subtitle.json \
--font ./fonts/Poppins-Bold.ttf \
--output ./alpha_overlays/ \
--width 1080 --height 1920
JSON 格式(由 audio_transcribe_lyrics 生成):
{
"segments": [
{"start": 0.0, "end": 2.5, "text": "Wanna make viral videos"},
...
]
}
"""
import argparse
import json
import sys
from pathlib import Path
from typing import Any, List, Union
import numpy as np
from PIL import Image, ImageDraw, ImageFont
# ─────────────────────────── 配置 ───────────────────────────
DEFAULT_STROKE_WIDTH = 3 # 标准字幕描边
LARGE_STROKE_WIDTH = 5 # 大号字幕描边
DEFAULT_FONT_SIZE = 48 # 标准字幕字号
LARGE_FONT_SIZE = 76 # 大号字幕字号
MAX_TEXT_WIDTH = 1020 # 超过此宽度自动缩小到 34pt
TEXT_COLOR = (255, 255, 255, 255) # 白色主体
STROKE_COLOR = (0, 0, 0, 255) # 黑色描边
WIDE_TEXT_WIDTH = 1020 # 超宽文本自动缩小的宽度阈值
WIDE_FONT_SIZE = 34 # 超宽文本自动缩小到的字号
# ─────────────────────────── 核心函数 ───────────────────────────
def load_font(font_path: str, size: int) -> Any:
"""加载字体,失败时回退到默认字体"""
try:
return ImageFont.truetype(font_path, size)
except Exception:
print(f"⚠️ 字体 {font_path} 加载失败,回退到默认字体")
return ImageFont.load_default()
def render_subtitle_frame(
text: str,
canvas_width: int,
canvas_height: int,
font: Any,
stroke_width: int,
y_position: int,
) -> Image.Image:
"""
渲染单条字幕为透明背景 PNG
Args:
text: 字幕文本
canvas_width: 画布宽度
canvas_height: 画布高度
font: PIL 字体对象
stroke_width: 描边宽度
y_position: 字幕 Y 轴起始位置
Returns:
PIL Image(RGBA)
"""
img = Image.new("RGBA", (canvas_width, canvas_height), (0, 0, 0, 0))
draw = ImageDraw.Draw(img)
# 计算文本尺寸
bbox = draw.textbbox((0, 0), text, font=font)
text_width = bbox[2] - bbox[0]
text_height = bbox[3] - bbox[1]
# 居中 X
x = (canvas_width - text_width) // 2
# 黑色描边(4 周)
for dx in range(-stroke_width, stroke_width + 1):
for dy in range(-stroke_width, stroke_width + 1):
if dx == 0 and dy == 0:
continue
draw.text((x + dx, y_position + dy), text, fill=STROKE_COLOR, font=font)
# 白色主体
draw.text((x, y_position), text, fill=TEXT_COLOR, font=font)
return img
def process_subtitles(
segments: List[dict],
canvas_width: int,
canvas_height: int,
font_path: str,
output_dir: Path,
) -> List[dict]:
"""
处理字幕列表,输出 PNG 文件序列
Returns:
带文件名的时间戳列表(可用于 ffmpeg overlay)
"""
output_dir.mkdir(parents=True, exist_ok=True)
# 自动检测大号字幕(用于标题/CTA)
def get_font_size(text: str, current_font_size: int) -> tuple:
bbox_estimate = (len(text) * current_font_size * 0.6, current_font_size)
if bbox_estimate[0] > MAX_TEXT_WIDTH:
return WIDE_FONT_SIZE, LARGE_STROKE_WIDTH
return current_font_size, DEFAULT_STROKE_WIDTH
timestamps = []
for i, seg in enumerate(segments):
text = seg["text"].strip()
start = seg["start"]
end = seg["end"]
# 检测是否为大号字幕(以感叹号结尾或全大写短句)
is_large = text.endswith("!") or (text.isupper() and len(text.split()) <= 5)
font_size = LARGE_FONT_SIZE if is_large else DEFAULT_FONT_SIZE
stroke = LARGE_STROKE_WIDTH if is_large else DEFAULT_STROKE_WIDTH
# 超宽文本自动缩小
if len(text) > 30:
font_size = min(font_size, WIDE_FONT_SIZE)
stroke = DEFAULT_STROKE_WIDTH
font = load_font(font_path, font_size)
# 字幕位置:数字人头顶上方(y ≈ 1050)
# 字幕高度约 60px(48pt),y=1050 时底部约 1110,不与 y=1227 的数字人重叠
y_position = 1050
img = render_subtitle_frame(
text, canvas_width, canvas_height, font, stroke, y_position
)
filename = f"sub_{i:02d}.png"
img.save(output_dir / filename, "PNG")
timestamps.append(
{
"index": i,
"filename": filename,
"start": start,
"end": end,
"text": text,
}
)
return timestamps
# ─────────────────────────── 主入口 ───────────────────────────
def main():
parser = argparse.ArgumentParser(description="字幕渲染:JSON 时间戳 → PNG 序列")
parser.add_argument(
"--script",
"-s",
required=True,
help="字幕时间戳 JSON 文件(由 audio_transcribe_lyrics 生成)",
)
parser.add_argument(
"--font", "-f", required=True, help="字体文件路径(推荐:Poppins-Bold.ttf)"
)
parser.add_argument("--output", "-o", required=True, help="输出目录(PNG 序列)")
parser.add_argument("--width", type=int, default=1080, help="画布宽度(默认 1080)")
parser.add_argument(
"--height", type=int, default=1920, help="画布高度(默认 1920)"
)
args = parser.parse_args()
# 加载 JSON
with open(args.script) as f:
data = json.load(f)
segments = data.get("segments", data)
if not segments:
print("❌ 未找到字幕 segments,请检查 JSON 格式")
sys.exit(1)
print(f"📝 处理 {len(segments)} 条字幕...")
output_dir = Path(args.output)
timestamps = process_subtitles(
segments=segments,
canvas_width=args.width,
canvas_height=args.height,
font_path=args.font,
output_dir=output_dir,
)
print(f"\n✅ 完成!输出 {len(timestamps)} 个 PNG → {output_dir}/")
print("ffmpeg overlay 命令示例:")
print(" ffmpeg -i main_video.mp4 -i sub_00.png \\")
print(
" -filter_complex \"[0:v][1:v]overlay=x=0:y=1050:enable='between(t,0.0,2.5)'\" \\"
)
print(" -c:a copy output.mp4")
if __name__ == "__main__":
main()