Add download_captions function with English preference and translation capabilities
This commit is contained in:
BIN
__pycache__/youtube_transcript_downloader.cpython-312.pyc
Normal file
BIN
__pycache__/youtube_transcript_downloader.cpython-312.pyc
Normal file
Binary file not shown.
285
youtube_transcript_downloader_backup.py
Normal file
285
youtube_transcript_downloader_backup.py
Normal file
@@ -0,0 +1,285 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
YouTube Transcript Downloader and Translator
|
||||
|
||||
This program downloads a YouTube video transcript and translates it to English.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import re
|
||||
from youtube_transcript_api import YouTubeTranscriptApi
|
||||
from youtube_transcript_api.formatters import TextFormatter
|
||||
from deep_translator import GoogleTranslator
|
||||
|
||||
|
||||
def extract_video_id(url):
|
||||
"""Extract YouTube video ID from various URL formats."""
|
||||
patterns = [
|
||||
r'(?:v=|\/)([0-9A-Za-z_-]{11}).*',
|
||||
r'youtu\.be\/([0-9A-Za-z_-]{11})',
|
||||
]
|
||||
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, url)
|
||||
if match:
|
||||
return match.group(1)
|
||||
return None
|
||||
|
||||
|
||||
def get_transcript(video_id, languages=None):
|
||||
"""
|
||||
Get transcript for a YouTube video.
|
||||
|
||||
Args:
|
||||
video_id: The YouTube video ID
|
||||
languages: List of language codes to try (default: ['en', 'en-US'])
|
||||
|
||||
Returns:
|
||||
Transcript object or None if not found
|
||||
"""
|
||||
if languages is None:
|
||||
languages = ['en', 'en-US']
|
||||
|
||||
try:
|
||||
transcript = YouTubeTranscriptApi().fetch(video_id, languages=languages)
|
||||
return transcript
|
||||
except Exception as e:
|
||||
print(f"Error getting transcript: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def get_transcript_text(transcript):
|
||||
"""Extract plain text from transcript."""
|
||||
formatter = TextFormatter()
|
||||
return formatter.format_transcript(transcript)
|
||||
|
||||
|
||||
def translate_text(text, target_language='en'):
|
||||
"""
|
||||
Translate text to target language using Google Translate.
|
||||
|
||||
Args:
|
||||
text: The text to translate
|
||||
target_language: Target language code (default: 'en')
|
||||
|
||||
Returns:
|
||||
Translated text or None if translation fails
|
||||
"""
|
||||
try:
|
||||
translator = GoogleTranslator(source='auto', target=target_language)
|
||||
translated = translator.translate(text)
|
||||
return translated
|
||||
except Exception as e:
|
||||
print(f"Error during translation: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def format_transcript(transcript, include_timestamps=True):
|
||||
"""
|
||||
Format transcript into readable text.
|
||||
|
||||
Args:
|
||||
transcript: The transcript list of dictionaries
|
||||
include_timestamps: Whether to include timestamps
|
||||
|
||||
Returns:
|
||||
Formatted string
|
||||
"""
|
||||
formatter = TextFormatter()
|
||||
return formatter.format_transcript(transcript)
|
||||
|
||||
|
||||
def detect_language(transcript):
|
||||
"""Attempt to detect the language of the transcript."""
|
||||
if not transcript:
|
||||
return "unknown"
|
||||
|
||||
# Get first few words to help identify language
|
||||
if isinstance(transcript, list) and len(transcript) > 0:
|
||||
text = transcript[0].get('text', '')
|
||||
return text[:100]
|
||||
return ""
|
||||
|
||||
|
||||
def main():
|
||||
"""Main function to run the transcript downloader."""
|
||||
print("YouTube Transcript Downloader and Translator")
|
||||
print("=" * 45)
|
||||
|
||||
# Get YouTube URL from command line argument or prompt user
|
||||
if len(sys.argv) > 1:
|
||||
url = sys.argv[1]
|
||||
else:
|
||||
url = input("Enter YouTube URL: ").strip()
|
||||
|
||||
# Extract video ID
|
||||
video_id = extract_video_id(url)
|
||||
if not video_id:
|
||||
print("Error: Could not extract video ID from URL")
|
||||
print("Please enter a valid YouTube URL")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Video ID: {video_id}")
|
||||
print("Fetching transcript...")
|
||||
|
||||
# Download captions with preference for English
|
||||
result = download_captions(video_id)
|
||||
|
||||
if not result['success']:
|
||||
print("\nCould not retrieve transcript.")
|
||||
print("Possible reasons:")
|
||||
print("- Video does not have captions/subtitles")
|
||||
print("- Video is live stream")
|
||||
print("- Video is private")
|
||||
sys.exit(1)
|
||||
|
||||
# Format original transcript
|
||||
original_formatted = format_transcript(result['transcript'], include_timestamps=True)
|
||||
|
||||
# Check if translation is needed (try to detect if it's already English)
|
||||
original_text = get_transcript_text(result['transcript'])
|
||||
|
||||
# Ask user if they want to translate
|
||||
print("\n" + "=" * 45)
|
||||
print("Transcript successfully retrieved!")
|
||||
print(f"Language: {result['language']}")
|
||||
|
||||
# Determine if translation is needed
|
||||
# If transcript is in English, no translation needed
|
||||
needs_translation = not is_english(original_text)
|
||||
|
||||
if needs_translation:
|
||||
print("\nNote: The transcript appears to be in a non-English language.")
|
||||
print("Would you like to translate it to English?")
|
||||
response = input("Enter 'y' for yes or 'n' for no: ").strip().lower()
|
||||
|
||||
if response == 'y':
|
||||
print("\nTranslating to English...")
|
||||
translated = translate_text(original_text, 'en')
|
||||
|
||||
if translated:
|
||||
# Save translated transcript
|
||||
output_file = f"transcript_{video_id}_translated.txt"
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
f.write("TRANSLATED TRANSCRIPT\n")
|
||||
f.write("=" * 45 + "\n\n")
|
||||
f.write(translated)
|
||||
|
||||
print(f"\nTranslated transcript saved to: {output_file}")
|
||||
|
||||
# Save both original and translated
|
||||
original_file = f"transcript_{video_id}_original.txt"
|
||||
with open(original_file, 'w', encoding='utf-8') as f:
|
||||
f.write(original_formatted)
|
||||
|
||||
print(f"Original transcript saved to: {original_file}")
|
||||
|
||||
# Display original and translated
|
||||
print("\n" + "=" * 45)
|
||||
print("ORIGINAL TRANSCRIPT:")
|
||||
print("=" * 45)
|
||||
print(original_formatted)
|
||||
|
||||
print("\n" + "=" * 45)
|
||||
print("TRANSLATED TRANSCRIPT:")
|
||||
print("=" * 45)
|
||||
print(translated)
|
||||
else:
|
||||
print("\nTranslation failed. Saving original transcript...")
|
||||
output_file = f"transcript_{video_id}.txt"
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
f.write(original_formatted)
|
||||
print(f"Transcript saved to: {output_file}")
|
||||
else:
|
||||
output_file = f"transcript_{video_id}.txt"
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
f.write(original_formatted)
|
||||
print(f"\nTranscript saved to: {output_file}")
|
||||
else:
|
||||
# Already in English, just save it
|
||||
output_file = f"transcript_{video_id}.txt"
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
f.write(original_formatted)
|
||||
print(f"\nTranscript saved to: {output_file}")
|
||||
print("\n" + "=" * 45)
|
||||
print("TRANSCRIPT:")
|
||||
print("=" * 45)
|
||||
print(original_formatted)
|
||||
|
||||
|
||||
def is_english(text):
|
||||
"""Simple check to determine if text is likely English."""
|
||||
# Common English words to check against
|
||||
common_english_words = {
|
||||
'the', 'be', 'to', 'of', 'and', 'a', 'in', 'that', 'have', 'i',
|
||||
'it', 'for', 'not', 'on', 'with', 'he', 'as', 'you', 'do', 'at',
|
||||
'this', 'but', 'his', 'by', 'from', 'they', 'we', 'say', 'her',
|
||||
'she', 'or', 'an', 'will', 'my', 'one', 'all', 'would', 'there',
|
||||
'their', 'what', 'so', 'up', 'out', 'if', 'about', 'who', 'get',
|
||||
'which', 'go', 'me', 'when', 'make', 'can', 'like', 'time', 'no',
|
||||
'just', 'him', 'know', 'take', 'people', 'into', 'year', 'your',
|
||||
'good', 'some', 'could', 'them', 'see', 'other', 'than', 'then',
|
||||
'now', 'look', 'only', 'come', 'its', 'over', 'think', 'also',
|
||||
'back', 'after', 'use', 'two', 'how', 'our', 'work', 'first', 'well',
|
||||
'way', 'even', 'new', 'want', 'because', 'any', 'these', 'give', 'day'
|
||||
}
|
||||
|
||||
# Simple heuristic: count common English words
|
||||
words = set(text.lower().split())
|
||||
matches = sum(1 for word in words if word in common_english_words)
|
||||
|
||||
# If we find many common English words, it's likely English
|
||||
return matches > len(words) * 0.05 if words else False
|
||||
|
||||
|
||||
def download_captions(video_id):
|
||||
"""
|
||||
Download closed captions from a YouTube video, preferring English.
|
||||
|
||||
Args:
|
||||
video_id: The YouTube video ID
|
||||
|
||||
Returns:
|
||||
Dictionary with 'transcript', 'language', and 'success' keys,
|
||||
or None if no transcript is available
|
||||
"""
|
||||
try:
|
||||
# First, try to get English transcript (preferably en, en-US, en-GB)
|
||||
english_languages = ['en', 'en-US', 'en-GB']
|
||||
transcript = YouTubeTranscriptApi().fetch(video_id, languages=english_languages)
|
||||
|
||||
if transcript:
|
||||
return {
|
||||
'transcript': transcript,
|
||||
'language': transcript.language,
|
||||
'success': True
|
||||
}
|
||||
|
||||
# If no English transcript is available, try any language
|
||||
print("No English transcript found. Trying to get transcript in original language...")
|
||||
transcript = YouTubeTranscriptApi().fetch(video_id, languages=[])
|
||||
|
||||
if transcript:
|
||||
return {
|
||||
'transcript': transcript,
|
||||
'language': transcript.language,
|
||||
'success': True
|
||||
}
|
||||
|
||||
return {
|
||||
'transcript': None,
|
||||
'language': None,
|
||||
'success': False
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error downloading captions: {e}")
|
||||
return {
|
||||
'transcript': None,
|
||||
'language': None,
|
||||
'success': False
|
||||
}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user