875 lines
41 KiB
Python
875 lines
41 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""
|
|
스토리보드 → 완성 쇼츠 영상(mp4) 조립 [A방식: 정지 일러스트 + 모션]
|
|
=====================================================================
|
|
|
|
gen_storyboard.py 가 만든 스토리보드를 받아 장면별로:
|
|
1) Gemini 이미지로 일러스트 생성 (캐릭터 일관성 = 1번 컷을 레퍼런스로 물림)
|
|
2) Gemini TTS 로 그 장면 나레이션 음성 생성 (장면 길이 = 음성 길이)
|
|
3) ffmpeg 로 Ken Burns(zoompan 줌) + 자막(PNG 오버레이) + 음성 합성 → 세로 1080x1920 mp4
|
|
(Intel QSV 하드웨어 인코딩 자동 사용, 없으면 libx264 ultrafast)
|
|
|
|
이미지/음성은 output/ 에 장면별로 캐싱한다. 재실행 시 이미 있으면 재사용(재과금 X).
|
|
다시 그리고 싶으면 --force.
|
|
|
|
설치: pip install google-genai pillow imageio-ffmpeg (moviepy 불필요 — ffmpeg 직접 렌더)
|
|
실행:
|
|
python make_short.py # 최신 *_storyboard.json
|
|
python make_short.py --storyboard output/xxx_storyboard.json --voice Puck
|
|
python make_short.py --force # 이미지/음성 새로 생성
|
|
"""
|
|
|
|
import json
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
import wave
|
|
from pathlib import Path
|
|
|
|
from google.genai import types
|
|
from PIL import Image
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
import media as MU
|
|
|
|
HERE = Path(__file__).parent
|
|
SHARED = HERE.parent # generator/ (fonts·bgm 공용 자산)
|
|
OUT_DIR = HERE.parent / "output" / "animation" # 통합 출력: shorts_all/output/animation/
|
|
W, H = 1080, 1920 # 세로 쇼츠
|
|
BAND_RATIO = 0.18 # 상단 흰 자막 밴드 높이 비율(그림은 그 아래, 안 가림)
|
|
NARR_SIZE = 48 # 나레이션 자막 글자 크기(고정 — 장면마다 안 변함)
|
|
BRAND_TITLE = "썰록" # 상단 밴드 제목
|
|
IMG_MODEL = "gemini-2.5-flash-image"
|
|
VISION_MODEL = "gemini-2.5-flash" # 실제 사진 내용 분석 + 장면 매칭용
|
|
TTS_MODEL = "gemini-2.5-pro-preview-tts" # 최상위 TTS(발음 품질 우선). 저렴하게: gemini-3.1-flash-tts-preview
|
|
TTS_VOICE = "Puck"
|
|
TTS_STYLE = ("다음 한 문장을 한국 숏폼 병맛 나레이션처럼 신나고 빠르게, "
|
|
"능청스럽게 약간 오버해서 읽어줘:")
|
|
# 한글 자막용 폰트 — 둥글둥글 B급 느낌의 '주아체'(fonts/ 에 동봉), 없으면 맑은고딕 볼드
|
|
FONT = str(SHARED / "fonts" / "Jua-Regular.ttf")
|
|
FONT_FALLBACK = r"C:\Windows\Fonts\malgunbd.ttf"
|
|
# 상단 제목('썰록') 전용 폰트 — 조선 느낌 명조(송명체). 자막과 일부러 다르게.
|
|
TITLE_FONT = str(SHARED / "fonts" / "SongMyung-Regular.ttf")
|
|
TITLE_SIZE = 58
|
|
|
|
# AI 이미지 모델은 글자(특히 한글)를 제대로 못 그려 간판/메뉴 글씨가 깨진다.
|
|
# → 모든 이미지 생성에서 '글자 자체를 그리지 말라'고 강하게 지시. (가게 이름은 자막/카드로 표기)
|
|
NO_TEXT = (" CRITICAL: Render NO text of any kind in the image — no letters, words, numbers, "
|
|
"Korean hangul, logos, or signage text. Keep every sign, banner, board and menu "
|
|
"completely BLANK with no writing on them.")
|
|
|
|
|
|
# ---------------------------------------------------------------- 데이터
|
|
IMG_EXTS = (".jpg", ".jpeg", ".png", ".webp")
|
|
|
|
|
|
def resolve_real_images(images_dir, real_scenes_arg):
|
|
"""--images 값 → 정렬된 실제 사진 경로 리스트, --real-scenes → 정수 리스트.
|
|
|
|
images_dir 해석 순서 (가게마다 load_image/<가게>/ 로 나눠 담는 걸 지원):
|
|
1) 절대경로면 그대로
|
|
2) 상대경로/이름이면 HERE/<값> (예: load_image/연산골막국수)
|
|
3) 그래도 없으면 HERE/load_image/<값> (예: --images 연산골막국수)
|
|
가장 먼저 '이미지가 들어있는' 폴더를 채택. 없으면 빈 리스트.
|
|
"""
|
|
real_images, used = [], None
|
|
if images_dir:
|
|
d = Path(images_dir)
|
|
candidates = [d] if d.is_absolute() else [HERE / d, HERE / "load_image" / d]
|
|
for c in candidates:
|
|
if c.exists() and c.is_dir():
|
|
imgs = sorted(p for p in c.iterdir() if p.suffix.lower() in IMG_EXTS)
|
|
if imgs:
|
|
real_images, used = imgs, c
|
|
break
|
|
if used:
|
|
print(f"■ 실제 사진 폴더: {used} ({len(real_images)}장)")
|
|
real_scenes = None
|
|
if real_scenes_arg:
|
|
real_scenes = [int(x) for x in re.split(r"[,\s]+", str(real_scenes_arg).strip()) if x]
|
|
return real_images, real_scenes
|
|
|
|
|
|
# 실제 사진 카드를 합성하는 장면에서, AI 배경이 카드 자리를 비워두도록 주는 포즈 힌트
|
|
PRESENT_POSES = [
|
|
"Place the cat king in the UPPER part of the frame, smiling and pointing down toward the "
|
|
"lower-center, leaving the lower-center area visually simple (a photo will be placed there).",
|
|
"Put the cat king on the LEFT side presenting with one paw to the right-center, keeping the "
|
|
"right-center area simple and uncluttered (a photo will be placed there).",
|
|
"Show the cat king at the BOTTOM looking up with sparkling eyes toward the upper-center, "
|
|
"keeping the upper-center area simple (a photo will be placed there).",
|
|
]
|
|
# PRESENT_POSES 순서에 맞춰 실제 사진 카드를 놓을 (가로,세로) 중심 비율
|
|
PRESENT_SPOTS = [(0.50, 0.62), (0.62, 0.50), (0.50, 0.40)]
|
|
|
|
# 고양이 왕의 '눈' 그림체를 귀엽게 유지하되, 눈을 '떴을 때'의 모양만 고정한다.
|
|
# (모든 장면을 동그란 뜬눈으로 강제하면 감김/실눈/윙크 같은 표정 다양성이 사라짐)
|
|
EYE_STYLE = (
|
|
"When the cat's eyes are open, they are the same cute friendly eyes: large round eyes "
|
|
"with big round solid black pupils and a small soft white highlight, gentle innocent "
|
|
"look. Keep this cute rounded eye style; never sharp, angry, fierce, glaring or "
|
|
"realistic predatory eyes."
|
|
)
|
|
|
|
# 눈을 감거나(감김/실눈/윙크/졸림/웃음·울음으로 눈이 변하는) 장면용: 표정은 장면대로 두되 화풍만 유지.
|
|
EYE_FREE = (
|
|
"In this scene the cat's eye EXPRESSION follows the moment — the eyes may be gently "
|
|
"closed, squinting, blinking, winking, teary or sleepy half-closed as appropriate. "
|
|
"Keep the same cute, soft, rounded drawing style (never sharp, angry or realistic), "
|
|
"but do NOT force wide-open round eyes here."
|
|
)
|
|
|
|
# image_prompt(영어)에 아래 표현이 있으면 '눈 표정 장면'으로 보고 뜬눈 고정/레퍼런스를 푼다.
|
|
_EYE_EXPR_KW = (
|
|
"closed eye", "eyes closed", "close its eye", "close his eye", "shut eye",
|
|
"shut its eye", "squint", "narrowed eye", "wink", "half-closed", "half closed",
|
|
"sleepy", "sleeping", "asleep", "dozing", "doze", "napping", "yawn", "drowsy",
|
|
"crying", "sob", "weeping", "tears", "teary", "laughing", "giggl", "chuckl",
|
|
"wince", "grimac", "blink",
|
|
)
|
|
|
|
|
|
def scene_needs_eye_expression(sc) -> bool:
|
|
"""장면이 눈 감김/실눈/윙크 등 '뜬눈 아닌' 표정을 필요로 하면 True."""
|
|
text = ((sc.get("image_prompt") or "") + " " + (sc.get("caption") or "")).lower()
|
|
return any(k in text for k in _EYE_EXPR_KW)
|
|
|
|
# 캐릭터 공식 레퍼런스: 이 폴더에 이미지를 넣어두면(예: king_ref.png),
|
|
# 모든 영상의 모든 컷이 그 고양이 왕의 얼굴/눈/복장/화풍을 그대로 따라간다.
|
|
CHAR_REF_DIR = HERE / "character"
|
|
|
|
|
|
def load_char_ref():
|
|
"""Shorts/character/ 안의 첫 이미지 바이트 반환(없으면 None). 모든 영상의 캐릭터 앵커로 쓰인다."""
|
|
if not CHAR_REF_DIR.is_dir():
|
|
return None
|
|
for p in sorted(CHAR_REF_DIR.iterdir()):
|
|
if p.suffix.lower() in IMG_EXTS:
|
|
return p.read_bytes()
|
|
return None
|
|
|
|
|
|
# ---------------------------------------------------------------- 이미지
|
|
def gen_image(client, prompt, ref_bytes=None, model=None, no_text=True,
|
|
ref_mode="full") -> bytes:
|
|
"""Gemini 이미지 생성. ref_bytes 주면 캐릭터 일관성 위해 레퍼런스로 사용.
|
|
|
|
model: 사용할 이미지 모델 (없으면 IMG_MODEL). 신형일수록 글자 렌더링 우수.
|
|
no_text: True 면 '그림에 글자 금지' 지시를 붙인다(글자 깨짐 방지).
|
|
ref_mode: 'full' = 레퍼런스의 캐릭터/복장/화풍을 통째로 유지.
|
|
'eyes' = 레퍼런스에선 '눈/표정만' 복제하고 복장·장면은 프롬프트대로.
|
|
"""
|
|
parts = []
|
|
if ref_bytes:
|
|
parts.append(types.Part.from_bytes(data=ref_bytes, mime_type="image/png"))
|
|
if ref_mode == "eyes":
|
|
prompt = ("Use the reference image ONLY to copy the cat character's EYES and "
|
|
"facial expression exactly: the same large round eyes with big round "
|
|
"solid black pupils and a small soft white highlight, gentle innocent "
|
|
"look. Do NOT copy the outfit, hat, robe color, pose, framing or "
|
|
"background from the reference — those follow the scene description. "
|
|
"New scene: " + prompt)
|
|
else:
|
|
prompt = ("Keep the EXACT same character design, outfit and art style as the "
|
|
"reference image. New scene: " + prompt)
|
|
parts.append(types.Part.from_text(text=prompt + (NO_TEXT if no_text else "")))
|
|
resp = client.models.generate_content(
|
|
model=model or IMG_MODEL,
|
|
contents=parts,
|
|
config=types.GenerateContentConfig(response_modalities=["IMAGE"]),
|
|
)
|
|
for p in resp.candidates[0].content.parts:
|
|
if getattr(p, "inline_data", None):
|
|
return p.inline_data.data
|
|
raise RuntimeError("이미지 파트 없음")
|
|
|
|
|
|
def composite_card(base_path, photo_path, out_path,
|
|
rel_w=0.60, center=(0.5, 0.60), angle=4.0, border=22):
|
|
"""AI 배경(base) 위에 '실제 사진'을 작은 폴라로이드 카드로 얹어 합성 저장.
|
|
|
|
사진은 원본 그대로(왜곡 X) 축소만 하고, 흰 테두리+그림자+살짝 기울기로 자연스럽게 올린다.
|
|
rel_w: 카드 사진 가로 = 화면폭 * rel_w. center: 카드 중심 (가로,세로) 비율. angle: 기울기(도).
|
|
"""
|
|
from PIL import Image, ImageFilter
|
|
|
|
base = Image.open(base_path).convert("RGBA")
|
|
photo = Image.open(photo_path).convert("RGB")
|
|
|
|
cw = int(W * rel_w)
|
|
ph = int(photo.height * (cw / photo.width))
|
|
max_h = int(H * 0.40) # 너무 길면 높이 기준으로 제한
|
|
if ph > max_h:
|
|
ph, cw = max_h, int(photo.width * (max_h / photo.height))
|
|
photo = photo.resize((cw, ph), Image.LANCZOS)
|
|
|
|
# 폴라로이드 흰 카드 (아래쪽 여백 좀 더)
|
|
card = Image.new("RGBA", (cw + 2 * border, ph + 2 * border + int(border * 1.4)),
|
|
(255, 255, 255, 255))
|
|
card.paste(photo, (border, border))
|
|
card = card.rotate(angle, expand=True, resample=Image.BICUBIC)
|
|
|
|
cx, cy = int(W * center[0]), int(H * center[1])
|
|
x, y = cx - card.width // 2, cy - card.height // 2
|
|
|
|
# 그림자
|
|
alpha = card.split()[3]
|
|
shadow = Image.new("RGBA", card.size, (0, 0, 0, 0))
|
|
shadow.paste(Image.new("RGBA", card.size, (0, 0, 0, 150)), (0, 0), alpha)
|
|
shadow = shadow.filter(ImageFilter.GaussianBlur(18))
|
|
base.alpha_composite(shadow, (x + 10, y + 16))
|
|
base.alpha_composite(card, (x, y))
|
|
base.convert("RGB").save(out_path)
|
|
|
|
|
|
def cover_crop(src: Path, dst: Path):
|
|
"""이미지를 1080x1920 꽉 차게 cover-crop 해서 저장."""
|
|
im = Image.open(src).convert("RGB")
|
|
scale = max(W / im.width, H / im.height)
|
|
nw, nh = int(im.width * scale + 0.5), int(im.height * scale + 0.5)
|
|
im = im.resize((nw, nh), Image.LANCZOS)
|
|
left, top = (nw - W) // 2, (nh - H) // 2
|
|
im.crop((left, top, left + W, top + H)).save(dst)
|
|
|
|
|
|
# ---------------------------------------------------------------- 음성
|
|
def _extract_audio(resp):
|
|
"""TTS 응답에서 (pcm, rate) 추출. 비었으면 None 반환(재시도 신호)."""
|
|
cands = getattr(resp, "candidates", None) or []
|
|
for c in cands:
|
|
content = getattr(c, "content", None)
|
|
parts = getattr(content, "parts", None) or [] if content else []
|
|
for p in parts:
|
|
inline = getattr(p, "inline_data", None)
|
|
if inline and inline.data:
|
|
m = re.search(r"rate=(\d+)", inline.mime_type or "")
|
|
return inline.data, (int(m.group(1)) if m else 24000)
|
|
return None
|
|
|
|
|
|
def gen_tts(client, text, voice, retries=4) -> bytes:
|
|
"""Gemini TTS. 프리뷰 모델이 가끔 빈 응답(content=None)을 주므로 재시도한다."""
|
|
last = ""
|
|
for attempt in range(1, retries + 1):
|
|
try:
|
|
resp = client.models.generate_content(
|
|
model=TTS_MODEL,
|
|
contents=f"{TTS_STYLE}\n\n{text}",
|
|
config=types.GenerateContentConfig(
|
|
response_modalities=["AUDIO"],
|
|
speech_config=types.SpeechConfig(
|
|
voice_config=types.VoiceConfig(
|
|
prebuilt_voice_config=types.PrebuiltVoiceConfig(voice_name=voice)
|
|
)
|
|
),
|
|
),
|
|
)
|
|
got = _extract_audio(resp)
|
|
if got:
|
|
return got
|
|
# 비었으면 왜 비었는지(finish_reason 등) 기록하고 재시도
|
|
fr = ""
|
|
try:
|
|
fr = str(getattr(resp.candidates[0], "finish_reason", "") or "")
|
|
except Exception:
|
|
fr = "no-candidates"
|
|
last = f"빈 응답(finish_reason={fr or '?'})"
|
|
except Exception as e:
|
|
last = f"{type(e).__name__}: {e}"
|
|
if attempt < retries:
|
|
wait = 2 * attempt # 2s, 4s, 6s … 점증 백오프
|
|
print(f"\n ↻ TTS 재시도 {attempt}/{retries-1} ({last}) — {wait}s 후",
|
|
end="", flush=True)
|
|
time.sleep(wait)
|
|
raise RuntimeError(f"TTS 실패(재시도 {retries}회): {last}\n 문장: {text[:40]}…")
|
|
|
|
|
|
def pcm_to_wav(pcm, path, rate):
|
|
with wave.open(str(path), "wb") as w:
|
|
w.setnchannels(1); w.setsampwidth(2); w.setframerate(rate)
|
|
w.writeframes(pcm)
|
|
|
|
|
|
def wav_seconds(path: Path) -> float:
|
|
with wave.open(str(path), "rb") as w:
|
|
return w.getnframes() / w.getframerate()
|
|
|
|
|
|
def _atempo_chain(speed: float) -> list:
|
|
"""atempo 는 0.5~2.0 범위만 받으므로 큰 배율은 곱으로 쪼개 체이닝."""
|
|
chain, s = [], speed
|
|
while s > 2.0:
|
|
chain.append("atempo=2.0"); s /= 2.0
|
|
while s < 0.5:
|
|
chain.append("atempo=0.5"); s *= 2.0
|
|
chain.append(f"atempo={s:.4f}")
|
|
return chain
|
|
|
|
|
|
def process_audio(exe, src_wav, dst_wav, speed=1.0, trim_silence=True,
|
|
threshold="-38dB", keep=0.06):
|
|
"""컷편집용 음성 가공: 앞뒤 침묵 제거 + 말 속도 올리기.
|
|
|
|
- trim_silence: 앞/뒤 침묵을 잘라 컷이 탁탁 넘어가게(가운데 호흡은 유지).
|
|
silenceremove 로 앞을 자르고, areverse 로 뒤집어 다시 앞(=원래 뒤)을 자른 뒤 복원.
|
|
keep: 잘라낸 끝에 남길 여유(초). 0 이면 너무 칼같이 잘려 어색할 수 있어 약간 남긴다.
|
|
- speed: 1.0=원본, 1.3 이면 30% 빠르게(음정 유지). 길이도 그만큼 줄어든다.
|
|
가공 결과가 비거나 실패하면 원본을 그대로 복사해 안전하게 폴백.
|
|
"""
|
|
af = []
|
|
if trim_silence:
|
|
one = (f"silenceremove=start_periods=1:start_silence={keep}:"
|
|
f"start_threshold={threshold}:detection=peak")
|
|
af += [one, "areverse", one, "areverse"]
|
|
if speed and abs(speed - 1.0) > 1e-3:
|
|
af += _atempo_chain(speed)
|
|
if not af:
|
|
shutil.copyfile(src_wav, dst_wav); return
|
|
try:
|
|
_run([exe, "-y", "-i", str(src_wav), "-af", ",".join(af),
|
|
"-ar", "24000", "-ac", "1", str(dst_wav)])
|
|
if wav_seconds(dst_wav) < 0.15: # 과하게 잘렸으면 폴백
|
|
raise RuntimeError("가공 후 음성이 너무 짧음")
|
|
except (RuntimeError, OSError):
|
|
shutil.copyfile(src_wav, dst_wav)
|
|
|
|
|
|
# ---------------------------------------------------------------- 조립 (ffmpeg 직접 렌더)
|
|
FPS = 30
|
|
|
|
|
|
def _ffmpeg() -> str:
|
|
import imageio_ffmpeg
|
|
return imageio_ffmpeg.get_ffmpeg_exe()
|
|
|
|
|
|
def _run(cmd):
|
|
r = subprocess.run(cmd, capture_output=True, text=True, encoding="utf-8", errors="replace")
|
|
if r.returncode != 0:
|
|
raise RuntimeError("ffmpeg 실패:\n" + (r.stderr or "")[-1500:])
|
|
return r
|
|
|
|
|
|
def pick_encoder(exe) -> list:
|
|
"""Intel QSV 하드웨어 인코딩이 실제로 동작하면 그걸, 아니면 libx264 ultrafast."""
|
|
test = subprocess.run(
|
|
[exe, "-hide_banner", "-f", "lavfi", "-i", "color=c=black:s=128x128:d=1",
|
|
"-c:v", "h264_qsv", "-f", "null", "-"],
|
|
capture_output=True, text=True, errors="replace")
|
|
if test.returncode == 0:
|
|
print(" 인코더: h264_qsv (Intel 하드웨어)")
|
|
return ["-c:v", "h264_qsv", "-global_quality", "24", "-preset", "fast"]
|
|
print(" 인코더: libx264 (ultrafast)")
|
|
return ["-c:v", "libx264", "-preset", "ultrafast", "-crf", "23"]
|
|
|
|
|
|
def _wrap_lines(draw, text, font, max_w):
|
|
"""글자 단위로 픽셀 폭(max_w)에 맞춰 줄바꿈."""
|
|
lines, cur = [], ""
|
|
for ch in text:
|
|
if ch == "\n":
|
|
lines.append(cur); cur = ""; continue
|
|
if draw.textlength(cur + ch, font=font) <= max_w:
|
|
cur += ch
|
|
else:
|
|
lines.append(cur); cur = ch
|
|
if cur:
|
|
lines.append(cur)
|
|
return lines or [""]
|
|
|
|
|
|
def _blossom(d, cx, cy, r, color):
|
|
"""간단한 5장 꽃잎 벚꽃 장식(제목 양옆 무늬)."""
|
|
import math
|
|
pr = r * 0.62 # 꽃잎 반지름
|
|
for k in range(5):
|
|
a = math.radians(-90 + k * 72)
|
|
px, py = cx + r * 0.55 * math.cos(a), cy + r * 0.55 * math.sin(a)
|
|
d.ellipse([px - pr, py - pr, px + pr, py + pr], fill=color)
|
|
d.ellipse([cx - r * 0.32, cy - r * 0.32, cx + r * 0.32, cy + r * 0.32],
|
|
fill=(255, 238, 175, 255)) # 꽃심
|
|
|
|
|
|
def render_band_png(text, out_path, font_path, title=BRAND_TITLE):
|
|
"""상단 흰 밴드에 제목 + 나레이션(검정 글씨)을 그린 1080x1920 PNG.
|
|
밴드 아래는 투명 → 그 자리에 그림이 들어가 '그림을 가리지 않는' 만화컷 레이아웃."""
|
|
from PIL import ImageDraw, ImageFont
|
|
band_h = int(H * BAND_RATIO)
|
|
img = Image.new("RGBA", (W, H), (0, 0, 0, 0))
|
|
d = ImageDraw.Draw(img)
|
|
d.rectangle([0, 0, W, band_h], fill=(255, 255, 255, 255)) # 흰 밴드
|
|
d.line([(0, band_h - 2), (W, band_h - 2)], fill=(0, 0, 0, 38), width=2) # 옅은 구분선
|
|
|
|
# ── 제목: 조선 느낌 폰트(TITLE_FONT) + 양옆 벚꽃 + 아래 구분선(끝까지·불투명) ──
|
|
title_y = int(H * 0.012)
|
|
title_fp = TITLE_FONT if Path(TITLE_FONT).exists() else font_path
|
|
tf = ImageFont.truetype(title_fp, TITLE_SIZE)
|
|
cx = W / 2
|
|
tw = d.textlength(title, font=tf)
|
|
d.text((cx, title_y), title, font=tf, fill=(50, 50, 50, 255), anchor="ma")
|
|
|
|
by = title_y + TITLE_SIZE * 0.55 # 꽃 세로 중심 ≒ 글자 중앙
|
|
gap = 44
|
|
rose = (228, 150, 168, 255)
|
|
_blossom(d, cx - tw / 2 - gap, by, 17, rose)
|
|
_blossom(d, cx + tw / 2 + gap, by, 17, rose)
|
|
|
|
rule_y = title_y + TITLE_SIZE + 26 # 제목 아래 구분선: 화면 끝까지, 불투명
|
|
d.line([(0, rule_y), (W, rule_y)], fill=(60, 60, 60, 255), width=4)
|
|
|
|
# 나레이션: 글자 크기는 항상 고정(NARR_SIZE), 길면 줄바꿈만. 세로 가운데 정렬.
|
|
top = rule_y + 22
|
|
avail = band_h - top - 24
|
|
f = ImageFont.truetype(font_path, NARR_SIZE)
|
|
lines = _wrap_lines(d, text, f, W - 150)
|
|
lh = int(NARR_SIZE * 1.4)
|
|
y = top + max(0, (avail - len(lines) * lh) // 2)
|
|
for ln in lines:
|
|
d.text((W / 2, y), ln, font=f, fill=(20, 20, 20, 255), anchor="ma")
|
|
y += lh
|
|
img.save(out_path)
|
|
|
|
|
|
LINK_CTA = "링크는 고정댓글 확인" # ▼ 등 특수문자는 자막 폰트(주아체)에 없어 □로 깨짐 → 한글만
|
|
|
|
|
|
def render_link_card_png(base_img_path, link, out_path, font_path, store=None, cta=None):
|
|
"""배경 사진을 어둡게 깔고 가게 이름 + 안내문구 + 링크를 큼직하게 그린 1080x1920 PNG."""
|
|
from PIL import ImageDraw, ImageFont
|
|
base = Image.open(base_img_path).convert("RGB")
|
|
base = Image.blend(base, Image.new("RGB", base.size, (0, 0, 0)), 0.55)
|
|
d = ImageDraw.Draw(base)
|
|
|
|
def centered(text, y, size, fill, fit_one_line=False):
|
|
if fit_one_line: # 한 줄에 들어갈 때까지 폰트 축소
|
|
while size > 40 and d.textlength(text, font=ImageFont.truetype(font_path, size)) > W - 140:
|
|
size -= 4
|
|
f = ImageFont.truetype(font_path, size)
|
|
for ln in _wrap_lines(d, text, f, W - 140):
|
|
w = d.textlength(ln, font=f)
|
|
d.text(((W - w) / 2, y), ln, font=f, fill=fill,
|
|
stroke_width=5, stroke_fill="black")
|
|
y += int(size * 1.25)
|
|
|
|
if store:
|
|
centered(store, int(H * 0.33), 88, "white", fit_one_line=True)
|
|
centered(cta or LINK_CTA, int(H * 0.43), 58, "#FFFFFF", fit_one_line=True)
|
|
centered(link, int(H * 0.51), 72, "#FFE94A", fit_one_line=True)
|
|
base.save(out_path)
|
|
|
|
|
|
# 모든 세그먼트를 같은 규격으로 만들어 concat copy 가 되게 한다.
|
|
_AUDIO_ARGS = ["-c:a", "aac", "-b:a", "160k", "-ar", "44100", "-ac", "2"]
|
|
|
|
# ---------------------------------------------------------------- BGM
|
|
AUDIO_EXTS = (".mp3", ".wav", ".m4a", ".aac", ".ogg", ".flac", ".opus")
|
|
|
|
|
|
def resolve_bgm(bgm_arg):
|
|
"""--bgm 값 → 실제 음악 파일 경로(없으면 None).
|
|
|
|
해석 순서(가게/분위기별로 bgm/ 폴더에 나눠 담는 걸 지원):
|
|
1) 파일이면 그대로
|
|
2) 폴더면 그 안 음악 중 무작위 1곡 (병맛 BGM 여러 개 넣어두고 매번 다르게)
|
|
3) 이름만 주면 HERE/<값> → HERE/bgm/<값> → 공용 generator/<값> → generator/bgm/<값> 순 탐색
|
|
"""
|
|
if not bgm_arg or str(bgm_arg).strip().lower() in ("none", "off", "no"):
|
|
return None
|
|
import random
|
|
p = Path(bgm_arg)
|
|
cands = [p] if p.is_absolute() else [HERE / p, HERE / "bgm" / p, SHARED / p, SHARED / "bgm" / p]
|
|
for c in cands:
|
|
if c.is_file():
|
|
return c
|
|
if c.is_dir():
|
|
files = [f for f in sorted(c.iterdir()) if f.suffix.lower() in AUDIO_EXTS]
|
|
if files:
|
|
return random.choice(files)
|
|
return None
|
|
|
|
|
|
def mix_bgm(exe, video_path, bgm_path, out_path, volume=0.3, fade=1.0):
|
|
"""완성 영상(나레이션 음성 포함) 위에 BGM 한 트랙을 작은 볼륨으로 깔아 최종본 저장.
|
|
|
|
- BGM 이 영상보다 짧으면 무한 루프(-stream_loop -1), 길면 영상 길이에 맞춰 잘림.
|
|
- 나레이션[0:a] + BGM[1:a]*volume 를 amix(duration=first)로 섞어 영상 길이에 맞춘다.
|
|
- 영상은 재인코딩하지 않는다(-c:v copy) → 빠름. 시작에 짧은 페이드인.
|
|
"""
|
|
fc = (f"[1:a]volume={volume},afade=t=in:st=0:d={fade}[bg];"
|
|
f"[0:a][bg]amix=inputs=2:duration=first:dropout_transition=0[a]")
|
|
_run([exe, "-y", "-i", str(video_path), "-stream_loop", "-1", "-i", str(bgm_path),
|
|
"-filter_complex", fc, "-map", "0:v", "-map", "[a]",
|
|
"-c:v", "copy", *_AUDIO_ARGS, "-shortest", "-movflags", "+faststart",
|
|
str(out_path)])
|
|
|
|
|
|
# Shared low-level media utilities. Keep local render.py for per-feature
|
|
# layout, prompts, local fonts, BGM folder resolution, and produce().
|
|
_ffmpeg = MU._ffmpeg
|
|
_run = MU._run
|
|
pick_encoder = MU.pick_encoder
|
|
pcm_to_wav = MU.pcm_to_wav
|
|
wav_seconds = MU.wav_seconds
|
|
process_audio = MU.process_audio
|
|
_wrap_lines = MU._wrap_lines
|
|
mix_bgm = MU.mix_bgm
|
|
|
|
|
|
def build_video(scenes, audio_paths, img_paths, out_path, link=None, link_seconds=3.0,
|
|
store=None, cta=None, speed=1.0, trim_silence=True, tail=0.15,
|
|
bgm=None, bgm_volume=0.3):
|
|
"""장면별 ffmpeg 세그먼트(Ken Burns 줌 + 자막 오버레이 + 음성) → concat 으로 최종 mp4.
|
|
|
|
speed/trim_silence: 컷편집 느낌 — 음성을 빠르게 + 앞뒤 침묵 제거(→ 컷이 타이트).
|
|
tail: 각 컷 음성 끝에 붙일 정적(초). 0 이면 말 끝나자마자 바로 다음 컷.
|
|
bgm: 배경음악 파일 경로. 주면 concat 후 BGM 을 작은 볼륨으로 깔아 한 번 더 믹스.
|
|
bgm_volume: BGM 볼륨(0~1). 기본 0.12 = 나레이션 안 묻히게 12%.
|
|
"""
|
|
exe = _ffmpeg()
|
|
enc = pick_encoder(exe)
|
|
font = FONT if Path(FONT).exists() else FONT_FALLBACK
|
|
out_path = Path(out_path)
|
|
tmp = out_path.parent / "_render"
|
|
tmp.mkdir(exist_ok=True)
|
|
segs = []
|
|
total = 0.0
|
|
|
|
for idx, (sc, ap, ip) in enumerate(zip(scenes, audio_paths, img_paths), 1):
|
|
# 컷편집용으로 음성 가공(빠르게 + 침묵 제거) 후 그 길이를 컷 길이로 사용
|
|
proc_ap = tmp / f"aud{idx:02d}.wav"
|
|
process_audio(exe, ap, proc_ap, speed=speed, trim_silence=trim_silence)
|
|
speech = wav_seconds(proc_ap)
|
|
dur = speech + max(0.0, tail) # 말 끝나고 약간의 여운만
|
|
frames = max(1, round(dur * FPS))
|
|
sub_png = tmp / f"sub{idx:02d}.png"
|
|
# 화면 자막 = 그 장면에서 '말하는 문장'(나레이션) 그대로 → 무음 시청자도 내용 전달
|
|
render_band_png(sc.get("narration") or sc.get("caption", ""), sub_png, font)
|
|
seg = tmp / f"seg{idx:02d}.mp4"
|
|
# 레이아웃: 흰 배경 위 → 그림은 상단 밴드 '아래' 영역에만(안 가림) → 밴드 PNG 오버레이
|
|
band_h = int(H * BAND_RATIO)
|
|
img_h = H - band_h
|
|
bw, bh = int(W * 1.2), int(img_h * 1.2) # 줌 여유용 1.2x
|
|
fc = (f"color=c=white:s={W}x{H}:r={FPS}[bg];"
|
|
f"[0:v]scale={bw}:{bh}:force_original_aspect_ratio=increase,"
|
|
f"crop={bw}:{bh},"
|
|
f"zoompan=z='min(zoom+0.0007,1.12)':d={frames}:"
|
|
f"x='iw/2-(iw/zoom/2)':y='ih/2-(ih/zoom/2)':s={W}x{img_h}:fps={FPS}[img];"
|
|
f"[bg][img]overlay=0:{band_h}[base];"
|
|
f"[base][1:v]overlay=0:0[v]")
|
|
# 음성 뒤에 tail 만큼 무음을 붙여 영상 길이와 맞춤(apad + -t 로 컷)
|
|
_run([exe, "-y", "-i", str(ip), "-i", str(sub_png), "-i", str(proc_ap),
|
|
"-filter_complex", fc, "-map", "[v]", "-map", "2:a",
|
|
"-af", "apad", "-t", f"{dur:.3f}", "-r", str(FPS), *enc, *_AUDIO_ARGS,
|
|
"-pix_fmt", "yuv420p", str(seg)])
|
|
segs.append(seg)
|
|
total += dur
|
|
print(f"[{idx:2d}] 세그먼트 (말 {speech:.1f}초 → 컷 {dur:.1f}초)")
|
|
print(f" ▶ 본편 합계 {total:.1f}초" + (f" + 링크 {link_seconds:.1f}초" if link else ""))
|
|
|
|
# 링크 카드 (무음)
|
|
if link and img_paths:
|
|
card_png = tmp / "linkcard.png"
|
|
render_link_card_png(img_paths[-1], link, card_png, font, store=store, cta=cta)
|
|
seg = tmp / "seg_link.mp4"
|
|
_run([exe, "-y", "-loop", "1", "-i", str(card_png),
|
|
"-f", "lavfi", "-i", "anullsrc=r=44100:cl=stereo",
|
|
"-t", f"{link_seconds:.3f}", "-r", str(FPS), *enc, *_AUDIO_ARGS,
|
|
"-pix_fmt", "yuv420p", str(seg)])
|
|
segs.append(seg)
|
|
print(f"[링크] 카드 세그먼트 ({link_seconds:.1f}초)")
|
|
|
|
# concat (재인코딩 없이 복사 → 즉시). 실패하면 재인코딩으로 폴백.
|
|
# BGM 이 있으면 일단 임시 본편으로 이어붙이고, 그 위에 BGM 을 믹스해 최종본을 만든다.
|
|
listf = tmp / "concat.txt"
|
|
# concat 목록은 절대경로로 (ffmpeg 가 목록파일 위치 기준으로 상대경로를 또 해석하는 문제 방지)
|
|
listf.write_text("".join(f"file '{s.resolve().as_posix()}'\n" for s in segs), encoding="utf-8")
|
|
concat_out = (tmp / "_body.mp4") if bgm else out_path
|
|
try:
|
|
_run([exe, "-y", "-f", "concat", "-safe", "0", "-i", str(listf),
|
|
"-c", "copy", "-movflags", "+faststart", str(concat_out)])
|
|
except RuntimeError:
|
|
_run([exe, "-y", "-f", "concat", "-safe", "0", "-i", str(listf),
|
|
*enc, *_AUDIO_ARGS, "-pix_fmt", "yuv420p", "-movflags", "+faststart",
|
|
str(concat_out)])
|
|
|
|
if bgm:
|
|
mix_bgm(exe, concat_out, bgm, out_path, volume=bgm_volume)
|
|
print(f" ♪ BGM 믹스: {Path(bgm).name} (볼륨 {bgm_volume})")
|
|
|
|
shutil.rmtree(tmp, ignore_errors=True)
|
|
|
|
|
|
def _mime_of(path) -> str:
|
|
return "image/jpeg" if Path(path).suffix.lower() in (".jpg", ".jpeg") else "image/png"
|
|
|
|
|
|
def analyze_photo(client, photo_path) -> str:
|
|
"""실제 사진 1장을 Gemini 비전으로 보고 짧은 한국어 라벨 반환 (예: '수육 한 접시')."""
|
|
data = Path(photo_path).read_bytes()
|
|
resp = client.models.generate_content(
|
|
model=VISION_MODEL,
|
|
contents=[
|
|
types.Part.from_bytes(data=data, mime_type=_mime_of(photo_path)),
|
|
types.Part.from_text(text=(
|
|
"이 사진의 핵심 대상을 한국어 명사구 12자 이내로만 답해. "
|
|
"음식이면 메뉴명(예: '수육', '메밀국수'), 가게면 '가게 외관'/'매장 내부', "
|
|
"메뉴판이면 '메뉴판'. 설명·문장 금지, 라벨만.")),
|
|
],
|
|
)
|
|
return (resp.text or "").strip().splitlines()[0][:20] if resp.text else ""
|
|
|
|
|
|
_MATCH_SCHEMA = {
|
|
"type": "object",
|
|
"properties": {
|
|
"assignments": {
|
|
"type": "array",
|
|
"items": {
|
|
"type": "object",
|
|
"properties": {
|
|
"photo_index": {"type": "integer"},
|
|
"scene_number": {"type": "integer"},
|
|
},
|
|
"required": ["photo_index", "scene_number"],
|
|
},
|
|
}
|
|
},
|
|
"required": ["assignments"],
|
|
}
|
|
|
|
|
|
def match_real_scenes(client, scenes, real_images, max_real=3) -> dict:
|
|
"""각 실제 사진을 비전 분석 → 내용이 가장 맞는 장면에 배치한 {장면번호: 사진경로} 반환.
|
|
|
|
실패하면 위치 기반(plan_real_scenes)으로 폴백한다.
|
|
"""
|
|
real_images = list(real_images or [])
|
|
if not real_images:
|
|
return {}
|
|
|
|
# 1) 사진별 내용 라벨
|
|
labels = []
|
|
for p in real_images:
|
|
try:
|
|
lab = analyze_photo(client, p)
|
|
except Exception:
|
|
lab = ""
|
|
labels.append(lab)
|
|
print(f" · {Path(p).name} → '{lab or '?'}'")
|
|
|
|
# 2) 라벨 ↔ 장면 매칭 (구조화 출력 1회)
|
|
scene_lines = "\n".join(
|
|
f"{i+1}. [{s.get('caption','')}] {s.get('narration','')}"
|
|
for i, s in enumerate(scenes))
|
|
photo_lines = "\n".join(f"{i}. {labels[i] or '내용불명'}" for i in range(len(real_images)))
|
|
prompt = (
|
|
"아래는 영상의 '장면 목록'과, 영상에 넣을 '실제 사진'들의 내용 라벨이다.\n"
|
|
"각 사진을 그 내용이 가장 잘 어울리는 장면에 배치하라.\n"
|
|
"[규칙]\n"
|
|
"- 한 장면엔 사진 하나, 한 사진은 한 장면에만.\n"
|
|
"- 그 음식/메뉴/가게를 직접 언급·소개하는 장면에 우선 배치.\n"
|
|
"- 비슷하면 마케팅/소개가 나오는 뒤쪽 장면을 선호.\n"
|
|
f"- 최대 {max_real}장까지만 배치(나머지는 제외).\n\n"
|
|
f"[장면 목록]\n{scene_lines}\n\n[사진 라벨]\n{photo_lines}\n")
|
|
try:
|
|
resp = client.models.generate_content(
|
|
model=VISION_MODEL,
|
|
contents=prompt,
|
|
config=types.GenerateContentConfig(
|
|
temperature=0.2,
|
|
response_mime_type="application/json",
|
|
response_schema=_MATCH_SCHEMA,
|
|
),
|
|
)
|
|
assigns = json.loads(resp.text).get("assignments", [])
|
|
except Exception as e:
|
|
print(f" (매칭 실패 → 위치 기반으로 폴백: {type(e).__name__})")
|
|
return plan_real_scenes(scenes, real_images, None, max_real)
|
|
|
|
real_map, used_photo = {}, set()
|
|
for a in assigns:
|
|
pi, sn = a.get("photo_index"), a.get("scene_number")
|
|
if pi is None or sn is None:
|
|
continue
|
|
if 0 <= pi < len(real_images) and 1 <= sn <= len(scenes) \
|
|
and pi not in used_photo and sn not in real_map:
|
|
real_map[sn] = Path(real_images[pi])
|
|
used_photo.add(pi)
|
|
if len(real_map) >= max_real:
|
|
break
|
|
return real_map or plan_real_scenes(scenes, real_images, None, max_real)
|
|
|
|
|
|
def plan_real_scenes(scenes, real_images, real_scenes=None, max_real=3) -> dict:
|
|
"""어느 장면 번호(1-based)에 어떤 실제 사진을 깔지 매핑 dict 로 반환.
|
|
|
|
- real_scenes 가 주어지면 그 장면 번호들에 사진을 순서대로 매핑.
|
|
- 없으면 '뒤쪽(마케팅 전환부) 장면들'에 사진 개수만큼(최대 max_real) 자동 배치.
|
|
"""
|
|
real_images = list(real_images or [])
|
|
real_map = {}
|
|
if not real_images:
|
|
return real_map
|
|
if real_scenes:
|
|
for no, img in zip(real_scenes, real_images):
|
|
if 1 <= no <= len(scenes):
|
|
real_map[no] = Path(img)
|
|
else:
|
|
n = min(len(real_images), max_real, len(scenes))
|
|
start = len(scenes) - n + 1 # 뒤에서 n개
|
|
for k in range(n):
|
|
real_map[start + k] = Path(real_images[k])
|
|
return real_map
|
|
|
|
|
|
def produce(client, sb, stem, workdir=None, voice=TTS_VOICE, force=False,
|
|
real_images=None, real_scenes=None, max_real=3, match_content=True,
|
|
link=None, link_seconds=3.0, store=None, cta=None,
|
|
img_model=None, allow_text=False,
|
|
speed=1.0, trim_silence=True, tail=0.15, workers=4,
|
|
bgm=None, bgm_volume=0.3) -> Path:
|
|
"""스토리보드 dict → 장면별 이미지/음성 생성 후 mp4 합성. 완성 경로 반환.
|
|
|
|
workdir: 이 작업의 결과/중간파일을 담을 폴더. 이미지·음성은 workdir/assets/ 에,
|
|
완성 mp4 는 workdir/{stem}.mp4 로 저장한다. (기본: output/{stem})
|
|
real_images: 실제 사진(가게/메뉴판 등) 경로 리스트. 뒤쪽 장면에 깔아 AI 컷과 섞는다.
|
|
real_scenes: 실제 사진을 깔 장면 번호(1-based) 리스트. 없으면 뒤쪽 자동.
|
|
max_real: 자동 배치 시 실제 사진으로 채울 최대 장면 수.
|
|
link: 주면 맨 마지막에 링크 카드 1장을 붙인다.
|
|
"""
|
|
scenes = sb["scenes"]
|
|
style = sb.get("style_guide", "")
|
|
char = sb.get("character_sheet", "")
|
|
workdir = Path(workdir) if workdir else (OUT_DIR / stem)
|
|
workdir.mkdir(parents=True, exist_ok=True)
|
|
assets = workdir / "assets"
|
|
assets.mkdir(exist_ok=True)
|
|
|
|
# 실제 사진 → 장면 배치 결정
|
|
# · real_scenes 수동지정이 최우선
|
|
# · 아니면 내용 매칭(비전). 단, 새로 생성할 장면이 있을 때만(전부 캐시면 비전 호출 낭비라 생략)
|
|
real_images = list(real_images or [])
|
|
# max_real 이 None/0/음수면 '폴더 사진 전부 사용'(장면 수 한도 내)
|
|
eff_max = max_real if (max_real and max_real > 0) else len(real_images)
|
|
need_gen = force or any(not (assets / f"scene{i:02d}.png").exists()
|
|
for i in range(1, len(scenes) + 1))
|
|
if not real_images:
|
|
real_map = {}
|
|
elif real_scenes:
|
|
real_map = plan_real_scenes(scenes, real_images, real_scenes, eff_max)
|
|
elif match_content and client is not None and need_gen:
|
|
print(f"■ 실제 사진 내용 분석(Gemini 비전) → 장면 매칭 (최대 {eff_max}장)")
|
|
real_map = match_real_scenes(client, scenes, real_images, eff_max)
|
|
else:
|
|
real_map = plan_real_scenes(scenes, real_images, None, eff_max)
|
|
real_order = {no: k for k, no in enumerate(sorted(real_map))} # 포즈/위치 번갈아 쓰기용
|
|
if real_map:
|
|
picks = ", ".join(f"{no}←{real_map[no].name}" for no in sorted(real_map))
|
|
print(f"■ 실제 사진 합성 장면: {picks}")
|
|
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
img_paths = [assets / f"scene{i:02d}.png" for i in range(1, len(scenes) + 1)]
|
|
audio_paths = [assets / f"scene{i:02d}.wav" for i in range(1, len(scenes) + 1)]
|
|
|
|
def make_image(i, ref, ref_mode="full"):
|
|
"""장면 i 이미지 1장 생성(필요시 실사진 합성). ref=캐릭터 앵커 바이트."""
|
|
sc = scenes[i - 1]
|
|
raw_png = assets / f"scene{i:02d}_raw.png"
|
|
img_png = assets / f"scene{i:02d}.png"
|
|
# 눈 표정 장면(감김/실눈/윙크/졸림/웃음·울음 등)은 뜬눈 고정을 풀고 표정대로 그린다.
|
|
# · '눈만 고정(eyes)' 레퍼런스도 이런 장면에선 빼야 눈이 안 떠짐.
|
|
free_eyes = scene_needs_eye_expression(sc)
|
|
eye = EYE_FREE if free_eyes else EYE_STYLE
|
|
use_ref = None if (free_eyes and ref_mode == "eyes") else ref
|
|
if i in real_map:
|
|
k = real_order[i]
|
|
pose = PRESENT_POSES[k % len(PRESENT_POSES)]
|
|
spot = PRESENT_SPOTS[k % len(PRESENT_SPOTS)]
|
|
prompt = (f"{style}. {char}. {eye} Scene: {sc['image_prompt']}. {pose} "
|
|
f"Vertical 9:16 composition.")
|
|
data = gen_image(client, prompt, ref_bytes=use_ref, model=img_model,
|
|
no_text=not allow_text, ref_mode=ref_mode)
|
|
raw_png.write_bytes(data)
|
|
cover_crop(raw_png, img_png) # 1) AI 배경
|
|
composite_card(img_png, real_map[i], img_png, center=spot) # 2) 실사진 카드
|
|
else:
|
|
prompt = f"{style}. {char}. {eye} Scene: {sc['image_prompt']}. Vertical 9:16 composition."
|
|
data = gen_image(client, prompt, ref_bytes=use_ref, model=img_model,
|
|
no_text=not allow_text, ref_mode=ref_mode)
|
|
raw_png.write_bytes(data)
|
|
cover_crop(raw_png, img_png)
|
|
|
|
# --- 이미지: 캐릭터 앵커(1컷) 먼저, 나머지는 병렬 ---
|
|
# 앵커 의존성: 모든 컷이 1번 컷(캐릭터)을 ref 로 참조해야 일관성이 유지된다.
|
|
# → 앵커 1장만 순차로 만든 뒤, 나머지는 동시에 생성(서로 독립).
|
|
need_img = [i for i in range(1, len(scenes) + 1)
|
|
if force or not img_paths[i - 1].exists()]
|
|
cached_img = [i for i in range(1, len(scenes) + 1) if i not in need_img]
|
|
for i in cached_img:
|
|
print(f"[{i:2d}] 이미지 캐시 사용")
|
|
|
|
# 공식 캐릭터 레퍼런스가 있으면 '눈만 고정' 모드: 모든 컷이 레퍼런스에서 눈/표정만 복제.
|
|
# 곤룡포·관모 등 복장은 복제하지 않고 왕별 character_sheet/장면 프롬프트대로 그려진다.
|
|
char_ref = load_char_ref()
|
|
if char_ref is not None:
|
|
print("■ 캐릭터 레퍼런스(눈만 고정) 사용: Shorts/character/ "
|
|
"→ 눈/표정은 통일, 복장은 왕별로 다르게")
|
|
ref_bytes, ref_mode, anchor = char_ref, "eyes", None
|
|
else:
|
|
ref_mode = "full"
|
|
ref_bytes = None
|
|
for i in range(1, len(scenes) + 1): # 캐시된 raw 가 있으면 앵커로 재사용
|
|
raw = assets / f"scene{i:02d}_raw.png"
|
|
if raw.exists() and not force:
|
|
ref_bytes = raw.read_bytes(); break
|
|
anchor = None
|
|
if ref_bytes is None:
|
|
anchor = next((i for i in need_img if i not in real_map), None)
|
|
if anchor is not None:
|
|
print(f"[{anchor:2d}] 캐릭터 앵커 이미지 생성...", flush=True)
|
|
make_image(anchor, None)
|
|
ref_bytes = (assets / f"scene{anchor:02d}_raw.png").read_bytes()
|
|
|
|
rest = [i for i in need_img if i != anchor]
|
|
if rest:
|
|
print(f"■ 이미지 {len(rest)}장 병렬 생성 (워커 {workers})...", flush=True)
|
|
with ThreadPoolExecutor(max_workers=workers) as ex:
|
|
futs = {ex.submit(make_image, i, ref_bytes, ref_mode): i for i in rest}
|
|
for fut in as_completed(futs):
|
|
i = futs[fut]; fut.result()
|
|
tag = f"실사합성({real_map[i].name})" if i in real_map else "이미지"
|
|
print(f"[{i:2d}] {tag} done", flush=True)
|
|
|
|
# --- 음성: 전부 독립 → 병렬 ---
|
|
def make_tts(i):
|
|
text = (scenes[i - 1].get("narration") or "").strip()
|
|
if not text:
|
|
# 빈 나레이션 방어: TTS 가 빈 텍스트에 빈 응답을 반복하다 죽으므로 0.3초 무음으로 대체.
|
|
# (정상 경로에선 gen_storyboard 가 빈 장면을 미리 거르지만, 만약을 대비한 이중 안전장치)
|
|
pcm_to_wav(b"\x00\x00" * int(24000 * 0.3), audio_paths[i - 1], 24000)
|
|
return
|
|
pcm, rate = gen_tts(client, text, voice)
|
|
pcm_to_wav(pcm, audio_paths[i - 1], rate)
|
|
|
|
need_wav = [i for i in range(1, len(scenes) + 1)
|
|
if force or not audio_paths[i - 1].exists()]
|
|
for i in (i for i in range(1, len(scenes) + 1) if i not in need_wav):
|
|
print(f" [{i:2d}] 음성 캐시 사용 ({wav_seconds(audio_paths[i - 1]):.1f}초)")
|
|
if need_wav:
|
|
print(f"■ 음성 {len(need_wav)}개 병렬 생성 (워커 {workers})...", flush=True)
|
|
with ThreadPoolExecutor(max_workers=workers) as ex:
|
|
futs = {ex.submit(make_tts, i): i for i in need_wav}
|
|
for fut in as_completed(futs):
|
|
i = futs[fut]; fut.result()
|
|
print(f" [{i:2d}] 음성 {wav_seconds(audio_paths[i - 1]):.1f}초", flush=True)
|
|
|
|
out = workdir / f"{stem}.mp4"
|
|
print(f"\n영상 합성 → {out.name}")
|
|
build_video(scenes, audio_paths, img_paths, out, link=link, link_seconds=link_seconds,
|
|
store=store, cta=cta, speed=speed, trim_silence=trim_silence, tail=tail,
|
|
bgm=bgm, bgm_volume=bgm_volume)
|
|
print(f"완성: {out}")
|
|
return out
|