# [メタ情報] # 識別子: 全自動で動画からvttファイルを生成する_exe # 補足: env終了 # [/メタ情報] 要約: このテキストは、長尺動画から全自動で高品質なWebVTT(.vtt)字幕を生成するPythonパイプラインスクリプトと、それをFinderから手軽に実行するためのAutomatorワークフローについて記述しています。 Pythonスクリプトは、FFmpegを用いて動画を無音区間や指定された秒数で精密に分割し、分割後のPTS(プレゼンテーションタイムスタンプ)を完全にリセットします。各分割動画はGoogle Gemini 3.6 Flash APIにアップロードされ、AIが高精度な文字起こしを行います。この際、システムプロンプトにより、日本語の不自然な空白除去、句読点の厳格な適用、フィラー除去、字幕のブロック時間管理、さらに単独話者か複数話者かに応じた話者ラベルの自動付与など、詳細な品質ルールが適用されます。API呼び出しの失敗時には自動リトライ機能も備わっています。 GeminiからのレスポンスはVTT形式で解析され、動画全体のタイムラインに合わせてタイムシフト補正後、すべてのブロックが結合され、連番が再付与されます。最終的なVTTファイルは元の動画と同じディレクトリに出力されます。 付属のAutomatorワークフローは、Finderで選択された動画ファイルに対し、指定されたPython仮想環境でこのスクリプトをTerminal経由で実行することで、GUIから字幕生成プロセスを自動的に開始できるよう設計されています。 /Users/XXXXXX/python_scripts/multimodal_vtt_pipeline.py ``` #!/usr/bin/env python3 # -*- coding: utf-8 -*- """ 長尺動画の全自動マルチモーダルVTT字幕生成パイプライン (FFmpeg精密分割・PTS完全リセット + Gemini 3.6 Flash + 低Temperatureフィラー除去 + タイムシフト自動結合 + 日本語空白自動整形) ※ 完全env適合版(APIキーおよびパスの外部隔離完了) ※ M2 Mac (XXXXXX) フロントエンド動作仕様 """ import os import sys import subprocess import shutil import re import time import tempfile import warnings import unicodedata from pathlib import Path from dotenv import load_dotenv warnings.filterwarnings("ignore", category=FutureWarning) from google import genai from google.genai import types # ========================================== # 1. 環境設定(金庫 .env からの動的ロード & フォールバック完全排除) # ========================================== ENV_PATH = Path.home() / "python_scripts" / ".env" load_dotenv(dotenv_path=ENV_PATH) API_KEY = os.getenv("GEMINI_API_KEY") # ===== 安全な起動時自律停止ガード ===== if not API_KEY: print("エラー:必要な鍵(GEMINI_API_KEY)が.envに設定されていません", file=sys.stderr) sys.exit(1) # ========================================== # 設定とシステムプロンプト # ========================================== SYSTEM_PROMPT = """添付された動画ファイルの音声を解析し、極めて正確な文字起こしを行い、指定されたWebVTT(.vtt)形式の構造(コードデータのみ)を出力してください。挨拶やマークダウン装飾、解説は一切出力しないでください。 [字幕構造ルール] 1行目: 通し番号(1から始まる整数) 2行目: タイムスタンプ(HH:MM:SS.mmm --> HH:MM:SS.mmm) 3行目: 日本語の字幕テキスト(文脈に応じた適切な句読点あり) 各ブロックの間には空行を1行入れる。 [日本語テキスト品質ルール(最重要)] - 単語と単語の間、助詞(は、が、の、を、に等)の前後に半角スペースや全角スペースを絶対に入れないでください。通常の自然な日本語として文字を詰めて記述してください。 - 句読点(、や。)の前後にスペースを入れないでください。 - 「えー」「あのー」「えっと」「まあ」「そのー」などの意味のないフィラー(言い淀み・無駄な発声)は自然に100%除去してください。 - 会話の抜け落ちがないよう、音声は一言一句漏らさず正確に文字起こししてください。 [タイムスタンプの厳格ルール] - タイムスタンプは必ず渡された動画ファイルの冒頭「00:00:00.000(0秒)」からの相対経過時間で記述してください。 - 前の字幕の終了時刻より後の字幕の開始時刻が逆戻り(逆行)しないようにしてください。 [話者区分の判断] 映像と音声を解析し、もし「複数人による会話(対談、インタビュー等)」であると判断した場合は、字幕テキストの先頭に必ず「[男性]:」「[女性]:」または「[話者A]:」「[話者B]:」などの話者ラベルを付与してください。 もし「単独の喋り手」による動画であると判断した場合は、話者ラベルは付与せず、通常の字幕テキストのみを出力してください。""" SPLIT_BASE_SECONDS = 270.0 # 基本分割秒数 (4分30秒) # ========================================== # ユーティリティ関数 # ========================================== def check_dependencies(): """必要なコマンドが設定されているか確認する""" if not shutil.which("ffmpeg") or not shutil.which("ffprobe"): print("エラー: FFmpeg または FFprobe がインストールされていません。Homebrew等でインストールしてください。", file=sys.stderr) sys.exit(1) def clean_japanese_spaces(text): """日本語文字間や句読点周りに混入した不自然なスペースを除去する""" jp_char = r'[\u3000-\u303F\u3040-\u309F\u30A0-\u30FF\u4E00-\u9FFF\uFF01-\uFF60]' # 日本語文字と日本語文字の間のスペースを削除 text = re.sub(f'({jp_char})[ \t]+({jp_char})', r'\1\2', text) text = re.sub(f'({jp_char})[ \t]+({jp_char})', r'\1\2', text) # 2回実行で連続スペース対策 # 句読点の前後のスペースを削除 text = re.sub(f'({jp_char})[ \t]+([、。,.])', r'\1\2', text) text = re.sub(f'([、。,.])[ \t]+({jp_char})', r'\1\2', text) return text def sec_to_time(sec_float): """秒数を HH:MM:SS.mmm フォーマットに変換""" m, s = divmod(sec_float, 60) h, m = divmod(m, 60) ms = int(round((s - int(s)) * 1000)) if ms >= 1000: ms = 0 s += 1 return f"{int(h):02d}:{int(m):02d}:{int(s):02d}.{ms:03d}" def get_video_duration(video_path): """FFprobe/FFmpegを使って動画の総再生時間(秒)を堅牢に取得(3段階マルチ解析)""" video_path = unicodedata.normalize('NFC', video_path) # 1. format=duration スキャン cmd1 = ['ffprobe', '-v', 'error', '-show_entries', 'format=duration', '-of', 'default=noprintwrappers=1:nokey=1', video_path] res1 = subprocess.run(cmd1, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) try: val = float(res1.stdout.strip()) if val > 0: return val except Exception: pass # 2. stream=duration スキャン cmd2 = ['ffprobe', '-v', 'error', '-select_streams', 'v:0', '-show_entries', 'stream=duration', '-of', 'default=noprintwrappers=1:nokey=1', video_path] res2 = subprocess.run(cmd2, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) try: val = float(res2.stdout.strip()) if val > 0: return val except Exception: pass # 3. ffmpeg -i 直接ヘッダーキャプチャ cmd3 = ['ffmpeg', '-i', video_path] res3 = subprocess.run(cmd3, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) match = re.search(r'Duration:\s*(\d+):(\d+):(\d+\.\d+)', res3.stderr) if match: h, m, s = float(match.group(1)), float(match.group(2)), float(match.group(3)) return h * 3600 + m * 60 + s print("警告: 動画の再生長を自動取得できませんでした。デフォルト値を使用します。", file=sys.stderr) return 0.0 def get_silence_points(video_path): """無音開始ポイント(秒)のリストを取得""" video_path = unicodedata.normalize('NFC', video_path) cmd = ['ffmpeg', '-i', video_path, '-af', 'silencedetect=noise=-30dB:d=1.0', '-f', 'null', '-'] result = subprocess.run(cmd, stderr=subprocess.PIPE, text=True) silence_points = [] for line in result.stderr.splitlines(): match = re.search(r'silence_start:\s*(\d+(\.\d+)?)', line) if match: silence_points.append(float(match.group(1))) return silence_points def parse_and_shift_vtt(vtt_text, offset_sec, part_duration): """VTT文字列を解析し、テキストの空白整形および安全補正を行ってブロックを返す""" vtt_text = clean_japanese_spaces(vtt_text) blocks = re.split(r'\n\s*\n', vtt_text.strip()) parsed_blocks = [] for b in blocks: lines = [line.strip() for line in b.splitlines() if line.strip()] stamp_idx = -1 for idx, line in enumerate(lines): if '-->' in line: stamp_idx = idx break if stamp_idx != -1: stamp_line = lines[stamp_idx] match = re.search(r'(\d+:\d+:\d+\.\d+|\d+:\d+\.\d+)\s*-->\s*(\d+:\d+:\d+\.\d+|\d+:\d+\.\d+)', stamp_line) if match: s_str, e_str = match.group(1), match.group(2) def to_sec(t_str): pts = t_str.split(':') if len(pts) == 3: return float(pts[0])*3600 + float(pts[1])*60 + float(pts[2]) else: return float(pts[0])*60 + float(pts[1]) start_sec = to_sec(s_str) + offset_sec end_sec = to_sec(e_str) + offset_sec # 分割パーツの物理範囲に収まるよう安全補正 max_allowed_end = offset_sec + part_duration + 5.0 if start_sec > max_allowed_end: continue if end_sec > max_allowed_end: end_sec = max_allowed_end text_lines = lines[stamp_idx+1:] text = "\n".join(text_lines) text = clean_japanese_spaces(text) if text: parsed_blocks.append({ 'start': start_sec, 'end': end_sec, 'text': text }) return parsed_blocks def process_video_part(client, part_path, offset_sec, part_duration, retry_count=3): """1つの動画パーツをGemini 3.6 Flash APIに送信して文字起こし(自動リトライ付き)""" print(f" 📤 動画パーツをアップロード中: {os.path.basename(part_path)}") uploaded_file = client.files.upload(file=part_path) # 処理完了を待機 while uploaded_file.state.name == "PROCESSING": time.sleep(2) uploaded_file = client.files.get(name=uploaded_file.name) if uploaded_file.state.name == "FAILED": raise Exception("Google API側での動画処理に失敗しました。") print(" 🤖 Gemini 3.6 Flash に高精度文字起こしをリクエスト中 (Temperature=0.1)...") for attempt in range(retry_count): try: response = client.models.generate_content( model='gemini-3.6-flash', contents=[uploaded_file, SYSTEM_PROMPT], config=types.GenerateContentConfig( temperature=0.1 ) ) vtt_raw = response.text vtt_cleaned = re.sub(r'^```vtt\s*', '', vtt_raw, flags=re.MULTILINE) vtt_cleaned = re.sub(r'^```\s*$', '', vtt_cleaned, flags=re.MULTILINE) blocks = parse_and_shift_vtt(vtt_cleaned, offset_sec, part_duration) if blocks: return blocks, uploaded_file else: print(f" ⚠️ レスポンスのパースに失敗。リトライします ({attempt+1}/{retry_count})...") except Exception as e: print(f" ⚠️ APIエラー ({e})。リトライします ({attempt+1}/{retry_count})...") time.sleep(3) raise Exception("規定のリトライ回数を超えましたが、有効な字幕データを取得できませんでした。") # ========================================== # メインパイプライン処理 # ========================================== def main(): check_dependencies() if len(sys.argv) < 2: print("使用方法: python multimodal_vtt_pipeline.py <動画ファイルのパス>", file=sys.stderr) sys.exit(1) raw_video_path = sys.argv[1] video_path = unicodedata.normalize('NFC', raw_video_path) if not os.path.exists(video_path): print(f"エラー: ファイルが見つかりません: {video_path}", file=sys.stderr) sys.exit(1) print(f"🎬 全自動VTT字幕生成を開始します: {os.path.basename(video_path)}") # 最新SDKのクライアント初期化 client = genai.Client(api_key=API_KEY) total_duration = get_video_duration(video_path) if total_duration <= 0.0: print("エラー: 動画の長さを取得できませんでした。", file=sys.stderr) sys.exit(1) print(f"⏱️ 動画の総再生時間: {sec_to_time(total_duration)}") # 分割ポイントの決定(無音区間の探索) silence_points = get_silence_points(video_path) split_points = [0.0] current_target = SPLIT_BASE_SECONDS while current_target < total_duration - 60.0: # 残り1分未満なら分割しない best_point = None min_diff = float('inf') for sp in silence_points: diff = abs(sp - current_target) if diff < min_diff and diff <= 45.0: # 目標値の前後45秒以内の無音を探す min_diff = diff best_point = sp if best_point: split_points.append(best_point) current_target = best_point + SPLIT_BASE_SECONDS else: split_points.append(current_target) current_target += SPLIT_BASE_SECONDS split_points.append(total_duration) all_vtt_blocks = [] # 一時フォルダを作成して分割処理 with tempfile.TemporaryDirectory() as tmp_dir: part_count = len(split_points) - 1 print(f"✂️ 動画を {part_count} 個に分割して処理します。") for i in range(part_count): start = split_points[i] end = split_points[i+1] part_dur = end - start ext = os.path.splitext(video_path)[1] part_path = os.path.join(tmp_dir, f"part_{i+1:03d}{ext}") print(f"\n[{i+1}/{part_count}] 無音点切り出し ({sec_to_time(start)} -> {sec_to_time(end)}, 判定長: {sec_to_time(part_dur)})") # FFmpegによる無再エンコード高速精密分割(PTSリセット付き) split_cmd = [ 'ffmpeg', '-y', '-ss', str(start), '-to', str(end), '-i', video_path, '-c', 'copy', '-avoid_negative_ts', 'make_zero', part_path ] subprocess.run(split_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) try: blocks, uploaded_file = process_video_part(client, part_path, start, part_dur) all_vtt_blocks.extend(blocks) except Exception as e: print(f"エラー発生 ([{i+1}/{part_count}]): {e}", file=sys.stderr) finally: if 'uploaded_file' in locals(): print(f"[{i+1}/{part_count}] API上の動画ファイルを削除中...") try: client.files.delete(name=uploaded_file.name) except: pass # 最終VTTの構築(通し番号の再付与と結合) print("\n📝 最終的なVTTを結合し、連番を再付与中...") dir_name = os.path.dirname(video_path) base_name = os.path.splitext(os.path.basename(video_path))[0] out_path = os.path.join(dir_name, f"{base_name}_completed.vtt") # 全体ソート(開始時刻順) all_vtt_blocks.sort(key=lambda x: x['start']) with open(out_path, 'w', encoding='utf-8') as f: f.write("WEBVTT\n\n") for idx, block in enumerate(all_vtt_blocks, 1): start_str = sec_to_time(block['start']) end_str = sec_to_time(block['end']) f.write(f"{idx}\n") f.write(f"{start_str} --> {end_str}\n") f.write(f"{block['text']}\n\n") print(f"✅ 完了! 保存先: {out_path}") try: subprocess.run([ "osascript", "-e", 'display notification "字幕作成が完了しました!" with title "XXXXXXシステム"' ]) except: pass if __name__ == "__main__": main() ``` Automator 全自動VTT字幕作成.workflow /Users/XXXXXX/Library/Services/全自動VTT字幕作成.workflow/ ファイルまたはフォルダ Finder.app ``` on run {input, parameters} -- ▼ Pythonスクリプトを保存したパス set pythonScript to "/Users/XXXXXX/python_scripts/multimodal_vtt_pipeline.py" -- ▼ 仮想環境(venv)のPython set pythonBin to "/Users/XXXXXX/venv_vtt/bin/python" repeat with f in input set videoPath to POSIX path of f set qVideoPath to quoted form of videoPath set qPythonScript to quoted form of pythonScript -- ターミナルで実行するコマンドを組み立てる set cmd to "export PATH=\"/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin\"; " & pythonBin & " " & qPythonScript & " " & qVideoPath -- ターミナルを起動してコマンドを実行(見える形) tell application "Terminal" activate do script cmd end tell end repeat return input end run ```