Initial commit through Cline Kanban
This commit is contained in:
74
README.md
Normal file
74
README.md
Normal file
@@ -0,0 +1,74 @@
|
||||
# YouTube Transcript Downloader and Translator
|
||||
|
||||
A Python program to download YouTube video transcripts and translate them to English.
|
||||
|
||||
## Features
|
||||
|
||||
- Extract YouTube video ID from various URL formats
|
||||
- Download video transcripts in multiple languages
|
||||
- Save transcripts to text files
|
||||
- Support for timestamped output
|
||||
- Automatic translation to English using Google Translate
|
||||
- Language detection to determine if translation is needed
|
||||
|
||||
## Installation
|
||||
|
||||
1. Install the required dependencies:
|
||||
|
||||
```bash
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
|
||||
Or install directly:
|
||||
|
||||
```bash
|
||||
pip install youtube-transcript-api deep-translator
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
### Command Line
|
||||
|
||||
```bash
|
||||
python youtube_transcript_downloader.py <youtube_url>
|
||||
```
|
||||
|
||||
### Interactive Mode
|
||||
|
||||
If no URL is provided, the program will prompt you to enter one:
|
||||
|
||||
```bash
|
||||
python youtube_transcript_downloader.py
|
||||
```
|
||||
|
||||
### Examples
|
||||
|
||||
```bash
|
||||
# Using a standard YouTube URL
|
||||
python youtube_transcript_downloader.py https://www.youtube.com/watch?v=dQw4w9WgXcQ
|
||||
|
||||
# Using a shortened youtu.be URL
|
||||
python youtube_transcript_downloader.py https://youtu.be/dQw4w9WgXcQ
|
||||
```
|
||||
|
||||
## Output
|
||||
|
||||
The program will:
|
||||
1. Display the transcript in the terminal
|
||||
2. Save the transcript to a file named `transcript_<video_id>.txt`
|
||||
3. If translation is needed, prompt the user to translate to English
|
||||
4. Save both original and translated transcripts to separate files
|
||||
|
||||
## Notes
|
||||
|
||||
- The video must have captions/subtitles available for transcript extraction
|
||||
- The program first tries to fetch English transcripts
|
||||
- If no English transcript is available, it will try to get the original language
|
||||
- The program automatically detects if translation is needed and prompts to translate to English
|
||||
- For videos without English captions, the program will prompt you to translate the transcript
|
||||
|
||||
## Requirements
|
||||
|
||||
- Python 3.6+
|
||||
- youtube-transcript-api
|
||||
- deep-translator
|
||||
2
requirements.txt
Normal file
2
requirements.txt
Normal file
@@ -0,0 +1,2 @@
|
||||
youtube-transcript-api>=0.6.0
|
||||
deep-translator>=1.1.0
|
||||
240
youtube_transcript_downloader.py
Normal file
240
youtube_transcript_downloader.py
Normal file
@@ -0,0 +1,240 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
YouTube Transcript Downloader and Translator
|
||||
|
||||
This program downloads a YouTube video transcript and translates it to English.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import re
|
||||
from youtube_transcript_api import YouTubeTranscriptApi
|
||||
from youtube_transcript_api.formatters import TextFormatter
|
||||
from deep_translator import GoogleTranslator
|
||||
|
||||
|
||||
def extract_video_id(url):
|
||||
"""Extract YouTube video ID from various URL formats."""
|
||||
patterns = [
|
||||
r'(?:v=|\/)([0-9A-Za-z_-]{11}).*',
|
||||
r'youtu\.be\/([0-9A-Za-z_-]{11})',
|
||||
]
|
||||
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, url)
|
||||
if match:
|
||||
return match.group(1)
|
||||
return None
|
||||
|
||||
|
||||
def get_transcript(video_id, languages=None):
|
||||
"""
|
||||
Get transcript for a YouTube video.
|
||||
|
||||
Args:
|
||||
video_id: The YouTube video ID
|
||||
languages: List of language codes to try (default: ['en', 'en-US'])
|
||||
|
||||
Returns:
|
||||
Transcript object or None if not found
|
||||
"""
|
||||
if languages is None:
|
||||
languages = ['en', 'en-US']
|
||||
|
||||
try:
|
||||
transcript = YouTubeTranscriptApi.get_transcript(video_id, languages=languages)
|
||||
return transcript
|
||||
except Exception as e:
|
||||
print(f"Error getting transcript: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def get_transcript_text(transcript):
|
||||
"""Extract plain text from transcript."""
|
||||
formatter = TextFormatter()
|
||||
return formatter.format_transcript(transcript)
|
||||
|
||||
|
||||
def translate_text(text, target_language='en'):
|
||||
"""
|
||||
Translate text to target language using Google Translate.
|
||||
|
||||
Args:
|
||||
text: The text to translate
|
||||
target_language: Target language code (default: 'en')
|
||||
|
||||
Returns:
|
||||
Translated text or None if translation fails
|
||||
"""
|
||||
try:
|
||||
translator = GoogleTranslator(source='auto', target=target_language)
|
||||
translated = translator.translate(text)
|
||||
return translated
|
||||
except Exception as e:
|
||||
print(f"Error during translation: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def format_transcript(transcript, include_timestamps=True):
|
||||
"""
|
||||
Format transcript into readable text.
|
||||
|
||||
Args:
|
||||
transcript: The transcript list of dictionaries
|
||||
include_timestamps: Whether to include timestamps
|
||||
|
||||
Returns:
|
||||
Formatted string
|
||||
"""
|
||||
formatter = TextFormatter()
|
||||
return formatter.format_transcript(transcript)
|
||||
|
||||
|
||||
def detect_language(transcript):
|
||||
"""Attempt to detect the language of the transcript."""
|
||||
if not transcript:
|
||||
return "unknown"
|
||||
|
||||
# Get first few words to help identify language
|
||||
if isinstance(transcript, list) and len(transcript) > 0:
|
||||
text = transcript[0].get('text', '')
|
||||
return text[:100]
|
||||
return ""
|
||||
|
||||
|
||||
def main():
|
||||
"""Main function to run the transcript downloader."""
|
||||
print("YouTube Transcript Downloader and Translator")
|
||||
print("=" * 45)
|
||||
|
||||
# Get YouTube URL from command line argument or prompt user
|
||||
if len(sys.argv) > 1:
|
||||
url = sys.argv[1]
|
||||
else:
|
||||
url = input("Enter YouTube URL: ").strip()
|
||||
|
||||
# Extract video ID
|
||||
video_id = extract_video_id(url)
|
||||
if not video_id:
|
||||
print("Error: Could not extract video ID from URL")
|
||||
print("Please enter a valid YouTube URL")
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Video ID: {video_id}")
|
||||
print("Fetching transcript...")
|
||||
|
||||
# Try to get English transcript first
|
||||
transcript = get_transcript(video_id, ['en', 'en-US', 'en-GB'])
|
||||
|
||||
if transcript is None:
|
||||
# Try to get transcript in any language
|
||||
print("\nNo English transcript found. Trying to get transcript in original language...")
|
||||
transcript = get_transcript(video_id, [])
|
||||
|
||||
if transcript is None:
|
||||
print("\nCould not retrieve transcript.")
|
||||
print("Possible reasons:")
|
||||
print("- Video does not have captions/subtitles")
|
||||
print("- Video is live stream")
|
||||
print("- Video is private")
|
||||
sys.exit(1)
|
||||
|
||||
# Format original transcript
|
||||
original_formatted = format_transcript(transcript, include_timestamps=True)
|
||||
|
||||
# Check if translation is needed (try to detect if it's already English)
|
||||
original_text = get_transcript_text(transcript)
|
||||
|
||||
# Ask user if they want to translate
|
||||
print("\n" + "=" * 45)
|
||||
print("Transcript successfully retrieved!")
|
||||
|
||||
# Determine if translation is needed
|
||||
# If transcript is in English, no translation needed
|
||||
needs_translation = not is_english(original_text)
|
||||
|
||||
if needs_translation:
|
||||
print("\nNote: The transcript appears to be in a non-English language.")
|
||||
print("Would you like to translate it to English?")
|
||||
response = input("Enter 'y' for yes or 'n' for no: ").strip().lower()
|
||||
|
||||
if response == 'y':
|
||||
print("\nTranslating to English...")
|
||||
translated = translate_text(original_text, 'en')
|
||||
|
||||
if translated:
|
||||
# Save translated transcript
|
||||
output_file = f"transcript_{video_id}_translated.txt"
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
f.write("TRANSLATED TRANSCRIPT\n")
|
||||
f.write("=" * 45 + "\n\n")
|
||||
f.write(translated)
|
||||
|
||||
print(f"\nTranslated transcript saved to: {output_file}")
|
||||
|
||||
# Save both original and translated
|
||||
original_file = f"transcript_{video_id}_original.txt"
|
||||
with open(original_file, 'w', encoding='utf-8') as f:
|
||||
f.write(original_formatted)
|
||||
|
||||
print(f"Original transcript saved to: {original_file}")
|
||||
|
||||
# Display original and translated
|
||||
print("\n" + "=" * 45)
|
||||
print("ORIGINAL TRANSCRIPT:")
|
||||
print("=" * 45)
|
||||
print(original_formatted)
|
||||
|
||||
print("\n" + "=" * 45)
|
||||
print("TRANSLATED TRANSCRIPT:")
|
||||
print("=" * 45)
|
||||
print(translated)
|
||||
else:
|
||||
print("\nTranslation failed. Saving original transcript...")
|
||||
output_file = f"transcript_{video_id}.txt"
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
f.write(original_formatted)
|
||||
print(f"Transcript saved to: {output_file}")
|
||||
else:
|
||||
output_file = f"transcript_{video_id}.txt"
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
f.write(original_formatted)
|
||||
print(f"\nTranscript saved to: {output_file}")
|
||||
else:
|
||||
# Already in English, just save it
|
||||
output_file = f"transcript_{video_id}.txt"
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
f.write(original_formatted)
|
||||
print(f"\nTranscript saved to: {output_file}")
|
||||
print("\n" + "=" * 45)
|
||||
print("TRANSCRIPT:")
|
||||
print("=" * 45)
|
||||
print(original_formatted)
|
||||
|
||||
|
||||
def is_english(text):
|
||||
"""Simple check to determine if text is likely English."""
|
||||
# Common English words to check against
|
||||
common_english_words = {
|
||||
'the', 'be', 'to', 'of', 'and', 'a', 'in', 'that', 'have', 'i',
|
||||
'it', 'for', 'not', 'on', 'with', 'he', 'as', 'you', 'do', 'at',
|
||||
'this', 'but', 'his', 'by', 'from', 'they', 'we', 'say', 'her',
|
||||
'she', 'or', 'an', 'will', 'my', 'one', 'all', 'would', 'there',
|
||||
'their', 'what', 'so', 'up', 'out', 'if', 'about', 'who', 'get',
|
||||
'which', 'go', 'me', 'when', 'make', 'can', 'like', 'time', 'no',
|
||||
'just', 'him', 'know', 'take', 'people', 'into', 'year', 'your',
|
||||
'good', 'some', 'could', 'them', 'see', 'other', 'than', 'then',
|
||||
'now', 'look', 'only', 'come', 'its', 'over', 'think', 'also',
|
||||
'back', 'after', 'use', 'two', 'how', 'our', 'work', 'first', 'well',
|
||||
'way', 'even', 'new', 'want', 'because', 'any', 'these', 'give', 'day'
|
||||
}
|
||||
|
||||
# Simple heuristic: count common English words
|
||||
words = set(text.lower().split())
|
||||
matches = sum(1 for word in words if word in common_english_words)
|
||||
|
||||
# If we find many common English words, it's likely English
|
||||
return matches > len(words) * 0.05 if words else False
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user