Files
Popiai-skill/social-media/hilo-promo/scripts/render_subtitles.py
T

220 lines
6.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
字幕渲染脚本 - 用 Python PIL 生成透明背景字幕 PNG
解决本地 ffmpeg 未编译 libfreetype/libfontconfig 导致 drawtext 不可用的问题
使用方式:
python render_subtitles.py --script voiceover_subtitle.json \
--font ./fonts/Poppins-Bold.ttf \
--output ./alpha_overlays/ \
--width 1080 --height 1920
JSON 格式(由 audio_transcribe_lyrics 生成):
{
"segments": [
{"start": 0.0, "end": 2.5, "text": "Wanna make viral videos"},
...
]
}
"""
import argparse
import json
import sys
from pathlib import Path
from typing import Any, List, Union
import numpy as np
from PIL import Image, ImageDraw, ImageFont
# ─────────────────────────── 配置 ───────────────────────────
DEFAULT_STROKE_WIDTH = 3 # 标准字幕描边
LARGE_STROKE_WIDTH = 5 # 大号字幕描边
DEFAULT_FONT_SIZE = 48 # 标准字幕字号
LARGE_FONT_SIZE = 76 # 大号字幕字号
MAX_TEXT_WIDTH = 1020 # 超过此宽度自动缩小到 34pt
TEXT_COLOR = (255, 255, 255, 255) # 白色主体
STROKE_COLOR = (0, 0, 0, 255) # 黑色描边
WIDE_TEXT_WIDTH = 1020 # 超宽文本自动缩小的宽度阈值
WIDE_FONT_SIZE = 34 # 超宽文本自动缩小到的字号
# ─────────────────────────── 核心函数 ───────────────────────────
def load_font(font_path: str, size: int) -> Any:
"""加载字体,失败时回退到默认字体"""
try:
return ImageFont.truetype(font_path, size)
except Exception:
print(f"⚠️ 字体 {font_path} 加载失败,回退到默认字体")
return ImageFont.load_default()
def render_subtitle_frame(
text: str,
canvas_width: int,
canvas_height: int,
font: Any,
stroke_width: int,
y_position: int,
) -> Image.Image:
"""
渲染单条字幕为透明背景 PNG
Args:
text: 字幕文本
canvas_width: 画布宽度
canvas_height: 画布高度
font: PIL 字体对象
stroke_width: 描边宽度
y_position: 字幕 Y 轴起始位置
Returns:
PIL ImageRGBA
"""
img = Image.new("RGBA", (canvas_width, canvas_height), (0, 0, 0, 0))
draw = ImageDraw.Draw(img)
# 计算文本尺寸
bbox = draw.textbbox((0, 0), text, font=font)
text_width = bbox[2] - bbox[0]
text_height = bbox[3] - bbox[1]
# 居中 X
x = (canvas_width - text_width) // 2
# 黑色描边(4 周)
for dx in range(-stroke_width, stroke_width + 1):
for dy in range(-stroke_width, stroke_width + 1):
if dx == 0 and dy == 0:
continue
draw.text((x + dx, y_position + dy), text, fill=STROKE_COLOR, font=font)
# 白色主体
draw.text((x, y_position), text, fill=TEXT_COLOR, font=font)
return img
def process_subtitles(
segments: List[dict],
canvas_width: int,
canvas_height: int,
font_path: str,
output_dir: Path,
) -> List[dict]:
"""
处理字幕列表,输出 PNG 文件序列
Returns:
带文件名的时间戳列表(可用于 ffmpeg overlay
"""
output_dir.mkdir(parents=True, exist_ok=True)
# 自动检测大号字幕(用于标题/CTA)
def get_font_size(text: str, current_font_size: int) -> tuple:
bbox_estimate = (len(text) * current_font_size * 0.6, current_font_size)
if bbox_estimate[0] > MAX_TEXT_WIDTH:
return WIDE_FONT_SIZE, LARGE_STROKE_WIDTH
return current_font_size, DEFAULT_STROKE_WIDTH
timestamps = []
for i, seg in enumerate(segments):
text = seg["text"].strip()
start = seg["start"]
end = seg["end"]
# 检测是否为大号字幕(以感叹号结尾或全大写短句)
is_large = text.endswith("!") or (text.isupper() and len(text.split()) <= 5)
font_size = LARGE_FONT_SIZE if is_large else DEFAULT_FONT_SIZE
stroke = LARGE_STROKE_WIDTH if is_large else DEFAULT_STROKE_WIDTH
# 超宽文本自动缩小
if len(text) > 30:
font_size = min(font_size, WIDE_FONT_SIZE)
stroke = DEFAULT_STROKE_WIDTH
font = load_font(font_path, font_size)
# 字幕位置:数字人头顶上方(y ≈ 1050)
# 字幕高度约 60px48pt),y=1050 时底部约 1110,不与 y=1227 的数字人重叠
y_position = 1050
img = render_subtitle_frame(
text, canvas_width, canvas_height, font, stroke, y_position
)
filename = f"sub_{i:02d}.png"
img.save(output_dir / filename, "PNG")
timestamps.append(
{
"index": i,
"filename": filename,
"start": start,
"end": end,
"text": text,
}
)
return timestamps
# ─────────────────────────── 主入口 ───────────────────────────
def main():
parser = argparse.ArgumentParser(description="字幕渲染:JSON 时间戳 → PNG 序列")
parser.add_argument(
"--script",
"-s",
required=True,
help="字幕时间戳 JSON 文件(由 audio_transcribe_lyrics 生成)",
)
parser.add_argument(
"--font", "-f", required=True, help="字体文件路径(推荐:Poppins-Bold.ttf"
)
parser.add_argument("--output", "-o", required=True, help="输出目录(PNG 序列)")
parser.add_argument("--width", type=int, default=1080, help="画布宽度(默认 1080")
parser.add_argument(
"--height", type=int, default=1920, help="画布高度(默认 1920"
)
args = parser.parse_args()
# 加载 JSON
with open(args.script) as f:
data = json.load(f)
segments = data.get("segments", data)
if not segments:
print("❌ 未找到字幕 segments,请检查 JSON 格式")
sys.exit(1)
print(f"📝 处理 {len(segments)} 条字幕...")
output_dir = Path(args.output)
timestamps = process_subtitles(
segments=segments,
canvas_width=args.width,
canvas_height=args.height,
font_path=args.font,
output_dir=output_dir,
)
print(f"\n✅ 完成!输出 {len(timestamps)} 个 PNG → {output_dir}/")
print("ffmpeg overlay 命令示例:")
print(" ffmpeg -i main_video.mp4 -i sub_00.png \\")
print(
" -filter_complex \"[0:v][1:v]overlay=x=0:y=1050:enable='between(t,0.0,2.5)'\" \\"
)
print(" -c:a copy output.mp4")
if __name__ == "__main__":
main()