# -*- coding: utf-8 -*- """ Best3 저수준 도구함 (TTS · 비전 · ffmpeg · BGM · 폰트) ====================================================== one_short / render 가 공통으로 쓰는 유틸 모음. 단독 실행용 아님. · 비전: analyze_photos(사진/클립프레임 라벨·점수), extract_frame · 음성: gen_tts(Gemini TTS), pcm_to_wav, wav_seconds, process_audio(무음제거·속도) · ffmpeg: _ffmpeg, _run, pick_encoder(QSV/libx264), _AUDIO_ARGS · 자막: _wrap_lines(픽셀폭 줄바꿈) · BGM: resolve_bgm(폴더 랜덤 1곡), mix_bgm 설치: pip install google-genai pillow imageio-ffmpeg """ import json import re import shutil import subprocess import time import wave from pathlib import Path from google.genai import types HERE = Path(__file__).parent W, H = 1080, 1920 # 세로 쇼츠 FPS = 30 VISION_MODEL = "gemini-2.5-flash" # 사진/클립 비전 분석용 TTS_MODEL = "gemini-2.5-flash-preview-tts" TTS_VOICE = "Aoede" # 산뜻하고 가벼운 보이스. 대안: Charon(차분) / Sulafat(따뜻) / Laomedeia(경쾌) TTS_STYLE = ("다음 문장을 밝고 경쾌하게, 적당히 빠른 속도로 활기차고 신나는 톤으로 읽어줘. " "지루하지 않게 텐션을 올리되, 과한 병맛·오버는 아니고 세련되게:") LINK_CTA = "소개한 곳 링크는 고정댓글에서 확인하세요" # info 카드 링크 안내 # 한글 자막용 폰트 — 주아체(fonts/ 동봉), 없으면 맑은고딕 볼드. 제목/정보용은 송명체. FONT = str(HERE / "fonts" / "Jua-Regular.ttf") FONT_FALLBACK = r"C:\Windows\Fonts\malgunbd.ttf" TITLE_FONT = str(HERE / "fonts" / "SongMyung-Regular.ttf") _AUDIO_ARGS = ["-c:a", "aac", "-b:a", "160k", "-ar", "44100", "-ac", "2"] # 수동 입력(링크 없이 직접 업로드)용 — 로컬 폴더의 이미지/영상 파일 목록 _LOCAL_IMG_EXTS = (".jpg", ".jpeg", ".png", ".webp", ".bmp") _LOCAL_VID_EXTS = (".mp4", ".mov", ".webm", ".m4v") def list_media(folder, kind="image"): """로컬 폴더의 이미지(kind='image') 또는 영상(kind='video') 파일을 정렬해 [Path...] 반환. 링크 없이 직접 사진/클립을 올릴 때 사용. 폴더가 없으면 빈 리스트.""" exts = _LOCAL_VID_EXTS if kind == "video" else _LOCAL_IMG_EXTS d = Path(folder) if folder else None if not d or not d.is_dir(): return [] return sorted(p for p in d.iterdir() if p.suffix.lower() in exts) # ---------------------------------------------------------------- ffmpeg def _ffmpeg() -> str: import imageio_ffmpeg return imageio_ffmpeg.get_ffmpeg_exe() def _run(cmd): r = subprocess.run(cmd, capture_output=True, text=True, encoding="utf-8", errors="replace") if r.returncode != 0: raise RuntimeError("ffmpeg 실패:\n" + (r.stderr or "")[-1500:]) return r def pick_encoder(exe) -> list: """Intel QSV 하드웨어 인코딩이 실제로 동작하면 그걸, 아니면 libx264 ultrafast.""" test = subprocess.run( [exe, "-hide_banner", "-f", "lavfi", "-i", "color=c=black:s=128x128:d=1", "-c:v", "h264_qsv", "-f", "null", "-"], capture_output=True, text=True, errors="replace") if test.returncode == 0: print(" 인코더: h264_qsv (Intel 하드웨어)") return ["-c:v", "h264_qsv", "-global_quality", "24", "-preset", "fast"] print(" 인코더: libx264 (ultrafast)") return ["-c:v", "libx264", "-preset", "ultrafast", "-crf", "23"] # ---------------------------------------------------------------- 비전 def analyze_photos(client, photo_paths, model=None): """사진/클립프레임을 비전으로 한 번에 분석. 반환 rows: 입력 순서대로 [{label, char_fit, has_text}]. label = 보이는 핵심 한국어 한 줄(나레이션 매칭용) char_fit = 컷이 매력적인 정도 1~10(클립·사진 순위용) has_text = 글자/자막/캡션이 박혀 보이면 True 실패 시 빈 라벨/0/False.""" n = len(photo_paths) parts = [types.Part.from_text(text=( "여러 이미지(가게 사진 또는 영상 캡처)다. 각 이미지를 분석해라.\n" "- label: 보이는 핵심을 짧은 한국어 한 줄로(예: '대게 한 상', '바다 노을 뷰', '야외 수영장', " "'아늑한 객실', '외관/간판'). 메뉴판이면 '메뉴판'.\n" "- char_fit: 광고 컷으로 매력적인 정도 1~10 " "(음식·음료·수영장·아늑한 실내=높게 / 메뉴판·로고·밋밋한 외관·사람 위주=낮게).\n" "- has_text: 이미지에 '글자/자막/캡션/큰 로고 텍스트'가 박혀 보이면 true, 깨끗하면 false.\n" "각 이미지 앞 [이미지 N] 의 N 을 index 로 써라."))] for i, p in enumerate(photo_paths): parts.append(types.Part.from_text(text=f"[이미지 {i}]")) parts.append(types.Part.from_bytes(data=Path(p).read_bytes(), mime_type="image/jpeg")) schema = {"type": "object", "properties": {"photos": {"type": "array", "items": { "type": "object", "properties": { "index": {"type": "integer"}, "label": {"type": "string"}, "char_fit": {"type": "integer"}, "has_text": {"type": "boolean"}}, "required": ["index", "label", "char_fit", "has_text"]}}}, "required": ["photos"]} resp = client.models.generate_content( model=model or VISION_MODEL, contents=parts, config=types.GenerateContentConfig( temperature=0.2, response_mime_type="application/json", response_schema=schema)) by = {r["index"]: r for r in json.loads(resp.text).get("photos", []) if 0 <= r.get("index", -1) < n} return [{"label": (by.get(i, {}).get("label") or "").strip(), "char_fit": by.get(i, {}).get("char_fit", 0), "has_text": bool(by.get(i, {}).get("has_text", False))} for i in range(n)] def extract_frame(video_path, out_jpg, at=1.0): """영상에서 프레임 1장 추출(비전 분석/썸네일용).""" _run([_ffmpeg(), "-y", "-ss", str(at), "-i", str(video_path), "-frames:v", "1", str(out_jpg)]) return out_jpg # ---------------------------------------------------------------- 음성 def _extract_audio(resp): """TTS 응답에서 (pcm, rate) 추출. 비었으면 None 반환(재시도 신호).""" cands = getattr(resp, "candidates", None) or [] for c in cands: content = getattr(c, "content", None) parts = getattr(content, "parts", None) or [] if content else [] for p in parts: inline = getattr(p, "inline_data", None) if inline and inline.data: m = re.search(r"rate=(\d+)", inline.mime_type or "") return inline.data, (int(m.group(1)) if m else 24000) return None def gen_tts(client, text, voice, retries=4) -> bytes: """Gemini TTS. 프리뷰 모델이 가끔 빈 응답(content=None)을 주므로 재시도한다.""" last = "" for attempt in range(1, retries + 1): try: resp = client.models.generate_content( model=TTS_MODEL, contents=f"{TTS_STYLE}\n\n{text}", config=types.GenerateContentConfig( response_modalities=["AUDIO"], speech_config=types.SpeechConfig( voice_config=types.VoiceConfig( prebuilt_voice_config=types.PrebuiltVoiceConfig(voice_name=voice) ) ), ), ) got = _extract_audio(resp) if got: return got # 비었으면 왜 비었는지(finish_reason 등) 기록하고 재시도 fr = "" try: fr = str(getattr(resp.candidates[0], "finish_reason", "") or "") except Exception: fr = "no-candidates" last = f"빈 응답(finish_reason={fr or '?'})" except Exception as e: last = f"{type(e).__name__}: {e}" if attempt < retries: wait = 2 * attempt # 2s, 4s, 6s … 점증 백오프 print(f"\n ↻ TTS 재시도 {attempt}/{retries-1} ({last}) — {wait}s 후", end="", flush=True) time.sleep(wait) raise RuntimeError(f"TTS 실패(재시도 {retries}회): {last}\n 문장: {text[:40]}…") def pcm_to_wav(pcm, path, rate): with wave.open(str(path), "wb") as w: w.setnchannels(1); w.setsampwidth(2); w.setframerate(rate) w.writeframes(pcm) def wav_seconds(path: Path) -> float: with wave.open(str(path), "rb") as w: return w.getnframes() / w.getframerate() def _atempo_chain(speed: float) -> list: """atempo 는 0.5~2.0 범위만 받으므로 큰 배율은 곱으로 쪼개 체이닝.""" chain, s = [], speed while s > 2.0: chain.append("atempo=2.0"); s /= 2.0 while s < 0.5: chain.append("atempo=0.5"); s *= 2.0 chain.append(f"atempo={s:.4f}") return chain def process_audio(exe, src_wav, dst_wav, speed=1.0, trim_silence=True, threshold="-38dB", keep=0.06): """컷편집용 음성 가공: 앞뒤 침묵 제거 + 말 속도 올리기. - trim_silence: 앞/뒤 침묵을 잘라 컷이 탁탁 넘어가게(가운데 호흡은 유지). silenceremove 로 앞을 자르고, areverse 로 뒤집어 다시 앞(=원래 뒤)을 자른 뒤 복원. keep: 잘라낸 끝에 남길 여유(초). 작을수록 마디가 바짝 붙는다. - speed: 1.0=원본, 1.3 이면 30% 빠르게(음정 유지). 길이도 그만큼 줄어든다. 가공 결과가 비거나 실패하면 원본을 그대로 복사해 안전하게 폴백. """ af = [] if trim_silence: one = (f"silenceremove=start_periods=1:start_silence={keep}:" f"start_threshold={threshold}:detection=peak") af += [one, "areverse", one, "areverse"] if speed and abs(speed - 1.0) > 1e-3: af += _atempo_chain(speed) if not af: shutil.copyfile(src_wav, dst_wav); return try: _run([exe, "-y", "-i", str(src_wav), "-af", ",".join(af), "-ar", "24000", "-ac", "1", str(dst_wav)]) if wav_seconds(dst_wav) < 0.15: # 과하게 잘렸으면 폴백 raise RuntimeError("가공 후 음성이 너무 짧음") except (RuntimeError, OSError): shutil.copyfile(src_wav, dst_wav) # ---------------------------------------------------------------- 자막 유틸 def _wrap_lines(draw, text, font, max_w): """글자 단위로 픽셀 폭(max_w)에 맞춰 줄바꿈.""" lines, cur = [], "" for ch in text: if ch == "\n": lines.append(cur); cur = ""; continue if draw.textlength(cur + ch, font=font) <= max_w: cur += ch else: lines.append(cur); cur = ch if cur: lines.append(cur) return lines or [""] # ---------------------------------------------------------------- BGM AUDIO_EXTS = (".mp3", ".wav", ".m4a", ".aac", ".ogg", ".flac", ".opus") def resolve_bgm(bgm_arg): """--bgm 값 → 실제 음악 파일 경로(없으면 None). 해석 순서(분위기별로 bgm/ 폴더에 나눠 담는 걸 지원): 1) 파일이면 그대로 2) 폴더면 그 안 음악 중 무작위 1곡 3) 이름만 주면 HERE/<값> → HERE/bgm/<값> 순으로 파일/폴더 탐색 """ if not bgm_arg or str(bgm_arg).strip().lower() in ("none", "off", "no"): return None import random p = Path(bgm_arg) cands = [p] if p.is_absolute() else [HERE / p, HERE / "bgm" / p] for c in cands: if c.is_file(): return c if c.is_dir(): files = [f for f in sorted(c.iterdir()) if f.suffix.lower() in AUDIO_EXTS] if files: return random.choice(files) return None def mix_bgm(exe, video_path, bgm_path, out_path, volume=0.3, fade=1.0): """완성 영상(나레이션 음성 포함) 위에 BGM 한 트랙을 작은 볼륨으로 깔아 최종본 저장. - BGM 이 영상보다 짧으면 무한 루프(-stream_loop -1), 길면 영상 길이에 맞춰 잘림. - 나레이션[0:a] + BGM[1:a]*volume 를 amix(duration=first)로 섞어 영상 길이에 맞춘다. - 영상은 재인코딩하지 않는다(-c:v copy) → 빠름. 시작에 짧은 페이드인. """ fc = (f"[1:a]volume={volume},afade=t=in:st=0:d={fade}[bg];" f"[0:a][bg]amix=inputs=2:duration=first:dropout_transition=0[a]") _run([exe, "-y", "-i", str(video_path), "-stream_loop", "-1", "-i", str(bgm_path), "-filter_complex", fc, "-map", "0:v", "-map", "[a]", "-c:v", "copy", *_AUDIO_ARGS, "-shortest", "-movflags", "+faststart", str(out_path)])