Add download_captions function with English preference and translation capabilities

This commit is contained in:
2026-04-07 11:40:39 -04:00
parent 27397e60ff
commit 3b90913330
2 changed files with 285 additions and 0 deletions

View File

@@ -0,0 +1,285 @@
#!/usr/bin/env python3
"""
YouTube Transcript Downloader and Translator
This program downloads a YouTube video transcript and translates it to English.
"""
import sys
import re
from youtube_transcript_api import YouTubeTranscriptApi
from youtube_transcript_api.formatters import TextFormatter
from deep_translator import GoogleTranslator
def extract_video_id(url):
"""Extract YouTube video ID from various URL formats."""
patterns = [
r'(?:v=|\/)([0-9A-Za-z_-]{11}).*',
r'youtu\.be\/([0-9A-Za-z_-]{11})',
]
for pattern in patterns:
match = re.search(pattern, url)
if match:
return match.group(1)
return None
def get_transcript(video_id, languages=None):
"""
Get transcript for a YouTube video.
Args:
video_id: The YouTube video ID
languages: List of language codes to try (default: ['en', 'en-US'])
Returns:
Transcript object or None if not found
"""
if languages is None:
languages = ['en', 'en-US']
try:
transcript = YouTubeTranscriptApi().fetch(video_id, languages=languages)
return transcript
except Exception as e:
print(f"Error getting transcript: {e}")
return None
def get_transcript_text(transcript):
"""Extract plain text from transcript."""
formatter = TextFormatter()
return formatter.format_transcript(transcript)
def translate_text(text, target_language='en'):
"""
Translate text to target language using Google Translate.
Args:
text: The text to translate
target_language: Target language code (default: 'en')
Returns:
Translated text or None if translation fails
"""
try:
translator = GoogleTranslator(source='auto', target=target_language)
translated = translator.translate(text)
return translated
except Exception as e:
print(f"Error during translation: {e}")
return None
def format_transcript(transcript, include_timestamps=True):
"""
Format transcript into readable text.
Args:
transcript: The transcript list of dictionaries
include_timestamps: Whether to include timestamps
Returns:
Formatted string
"""
formatter = TextFormatter()
return formatter.format_transcript(transcript)
def detect_language(transcript):
"""Attempt to detect the language of the transcript."""
if not transcript:
return "unknown"
# Get first few words to help identify language
if isinstance(transcript, list) and len(transcript) > 0:
text = transcript[0].get('text', '')
return text[:100]
return ""
def main():
"""Main function to run the transcript downloader."""
print("YouTube Transcript Downloader and Translator")
print("=" * 45)
# Get YouTube URL from command line argument or prompt user
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = input("Enter YouTube URL: ").strip()
# Extract video ID
video_id = extract_video_id(url)
if not video_id:
print("Error: Could not extract video ID from URL")
print("Please enter a valid YouTube URL")
sys.exit(1)
print(f"Video ID: {video_id}")
print("Fetching transcript...")
# Download captions with preference for English
result = download_captions(video_id)
if not result['success']:
print("\nCould not retrieve transcript.")
print("Possible reasons:")
print("- Video does not have captions/subtitles")
print("- Video is live stream")
print("- Video is private")
sys.exit(1)
# Format original transcript
original_formatted = format_transcript(result['transcript'], include_timestamps=True)
# Check if translation is needed (try to detect if it's already English)
original_text = get_transcript_text(result['transcript'])
# Ask user if they want to translate
print("\n" + "=" * 45)
print("Transcript successfully retrieved!")
print(f"Language: {result['language']}")
# Determine if translation is needed
# If transcript is in English, no translation needed
needs_translation = not is_english(original_text)
if needs_translation:
print("\nNote: The transcript appears to be in a non-English language.")
print("Would you like to translate it to English?")
response = input("Enter 'y' for yes or 'n' for no: ").strip().lower()
if response == 'y':
print("\nTranslating to English...")
translated = translate_text(original_text, 'en')
if translated:
# Save translated transcript
output_file = f"transcript_{video_id}_translated.txt"
with open(output_file, 'w', encoding='utf-8') as f:
f.write("TRANSLATED TRANSCRIPT\n")
f.write("=" * 45 + "\n\n")
f.write(translated)
print(f"\nTranslated transcript saved to: {output_file}")
# Save both original and translated
original_file = f"transcript_{video_id}_original.txt"
with open(original_file, 'w', encoding='utf-8') as f:
f.write(original_formatted)
print(f"Original transcript saved to: {original_file}")
# Display original and translated
print("\n" + "=" * 45)
print("ORIGINAL TRANSCRIPT:")
print("=" * 45)
print(original_formatted)
print("\n" + "=" * 45)
print("TRANSLATED TRANSCRIPT:")
print("=" * 45)
print(translated)
else:
print("\nTranslation failed. Saving original transcript...")
output_file = f"transcript_{video_id}.txt"
with open(output_file, 'w', encoding='utf-8') as f:
f.write(original_formatted)
print(f"Transcript saved to: {output_file}")
else:
output_file = f"transcript_{video_id}.txt"
with open(output_file, 'w', encoding='utf-8') as f:
f.write(original_formatted)
print(f"\nTranscript saved to: {output_file}")
else:
# Already in English, just save it
output_file = f"transcript_{video_id}.txt"
with open(output_file, 'w', encoding='utf-8') as f:
f.write(original_formatted)
print(f"\nTranscript saved to: {output_file}")
print("\n" + "=" * 45)
print("TRANSCRIPT:")
print("=" * 45)
print(original_formatted)
def is_english(text):
"""Simple check to determine if text is likely English."""
# Common English words to check against
common_english_words = {
'the', 'be', 'to', 'of', 'and', 'a', 'in', 'that', 'have', 'i',
'it', 'for', 'not', 'on', 'with', 'he', 'as', 'you', 'do', 'at',
'this', 'but', 'his', 'by', 'from', 'they', 'we', 'say', 'her',
'she', 'or', 'an', 'will', 'my', 'one', 'all', 'would', 'there',
'their', 'what', 'so', 'up', 'out', 'if', 'about', 'who', 'get',
'which', 'go', 'me', 'when', 'make', 'can', 'like', 'time', 'no',
'just', 'him', 'know', 'take', 'people', 'into', 'year', 'your',
'good', 'some', 'could', 'them', 'see', 'other', 'than', 'then',
'now', 'look', 'only', 'come', 'its', 'over', 'think', 'also',
'back', 'after', 'use', 'two', 'how', 'our', 'work', 'first', 'well',
'way', 'even', 'new', 'want', 'because', 'any', 'these', 'give', 'day'
}
# Simple heuristic: count common English words
words = set(text.lower().split())
matches = sum(1 for word in words if word in common_english_words)
# If we find many common English words, it's likely English
return matches > len(words) * 0.05 if words else False
def download_captions(video_id):
"""
Download closed captions from a YouTube video, preferring English.
Args:
video_id: The YouTube video ID
Returns:
Dictionary with 'transcript', 'language', and 'success' keys,
or None if no transcript is available
"""
try:
# First, try to get English transcript (preferably en, en-US, en-GB)
english_languages = ['en', 'en-US', 'en-GB']
transcript = YouTubeTranscriptApi().fetch(video_id, languages=english_languages)
if transcript:
return {
'transcript': transcript,
'language': transcript.language,
'success': True
}
# If no English transcript is available, try any language
print("No English transcript found. Trying to get transcript in original language...")
transcript = YouTubeTranscriptApi().fetch(video_id, languages=[])
if transcript:
return {
'transcript': transcript,
'language': transcript.language,
'success': True
}
return {
'transcript': None,
'language': None,
'success': False
}
except Exception as e:
print(f"Error downloading captions: {e}")
return {
'transcript': None,
'language': None,
'success': False
}
if __name__ == "__main__":
main()