Initial commit through Cline Kanban

This commit is contained in:
2026-04-07 10:04:27 -04:00
commit 656580df52
3 changed files with 316 additions and 0 deletions

74
README.md Normal file
View File

@@ -0,0 +1,74 @@
# YouTube Transcript Downloader and Translator
A Python program to download YouTube video transcripts and translate them to English.
## Features
- Extract YouTube video ID from various URL formats
- Download video transcripts in multiple languages
- Save transcripts to text files
- Support for timestamped output
- Automatic translation to English using Google Translate
- Language detection to determine if translation is needed
## Installation
1. Install the required dependencies:
```bash
pip install -r requirements.txt
```
Or install directly:
```bash
pip install youtube-transcript-api deep-translator
```
## Usage
### Command Line
```bash
python youtube_transcript_downloader.py <youtube_url>
```
### Interactive Mode
If no URL is provided, the program will prompt you to enter one:
```bash
python youtube_transcript_downloader.py
```
### Examples
```bash
# Using a standard YouTube URL
python youtube_transcript_downloader.py https://www.youtube.com/watch?v=dQw4w9WgXcQ
# Using a shortened youtu.be URL
python youtube_transcript_downloader.py https://youtu.be/dQw4w9WgXcQ
```
## Output
The program will:
1. Display the transcript in the terminal
2. Save the transcript to a file named `transcript_<video_id>.txt`
3. If translation is needed, prompt the user to translate to English
4. Save both original and translated transcripts to separate files
## Notes
- The video must have captions/subtitles available for transcript extraction
- The program first tries to fetch English transcripts
- If no English transcript is available, it will try to get the original language
- The program automatically detects if translation is needed and prompts to translate to English
- For videos without English captions, the program will prompt you to translate the transcript
## Requirements
- Python 3.6+
- youtube-transcript-api
- deep-translator

2
requirements.txt Normal file
View File

@@ -0,0 +1,2 @@
youtube-transcript-api>=0.6.0
deep-translator>=1.1.0

View File

@@ -0,0 +1,240 @@
#!/usr/bin/env python3
"""
YouTube Transcript Downloader and Translator
This program downloads a YouTube video transcript and translates it to English.
"""
import sys
import re
from youtube_transcript_api import YouTubeTranscriptApi
from youtube_transcript_api.formatters import TextFormatter
from deep_translator import GoogleTranslator
def extract_video_id(url):
"""Extract YouTube video ID from various URL formats."""
patterns = [
r'(?:v=|\/)([0-9A-Za-z_-]{11}).*',
r'youtu\.be\/([0-9A-Za-z_-]{11})',
]
for pattern in patterns:
match = re.search(pattern, url)
if match:
return match.group(1)
return None
def get_transcript(video_id, languages=None):
"""
Get transcript for a YouTube video.
Args:
video_id: The YouTube video ID
languages: List of language codes to try (default: ['en', 'en-US'])
Returns:
Transcript object or None if not found
"""
if languages is None:
languages = ['en', 'en-US']
try:
transcript = YouTubeTranscriptApi.get_transcript(video_id, languages=languages)
return transcript
except Exception as e:
print(f"Error getting transcript: {e}")
return None
def get_transcript_text(transcript):
"""Extract plain text from transcript."""
formatter = TextFormatter()
return formatter.format_transcript(transcript)
def translate_text(text, target_language='en'):
"""
Translate text to target language using Google Translate.
Args:
text: The text to translate
target_language: Target language code (default: 'en')
Returns:
Translated text or None if translation fails
"""
try:
translator = GoogleTranslator(source='auto', target=target_language)
translated = translator.translate(text)
return translated
except Exception as e:
print(f"Error during translation: {e}")
return None
def format_transcript(transcript, include_timestamps=True):
"""
Format transcript into readable text.
Args:
transcript: The transcript list of dictionaries
include_timestamps: Whether to include timestamps
Returns:
Formatted string
"""
formatter = TextFormatter()
return formatter.format_transcript(transcript)
def detect_language(transcript):
"""Attempt to detect the language of the transcript."""
if not transcript:
return "unknown"
# Get first few words to help identify language
if isinstance(transcript, list) and len(transcript) > 0:
text = transcript[0].get('text', '')
return text[:100]
return ""
def main():
"""Main function to run the transcript downloader."""
print("YouTube Transcript Downloader and Translator")
print("=" * 45)
# Get YouTube URL from command line argument or prompt user
if len(sys.argv) > 1:
url = sys.argv[1]
else:
url = input("Enter YouTube URL: ").strip()
# Extract video ID
video_id = extract_video_id(url)
if not video_id:
print("Error: Could not extract video ID from URL")
print("Please enter a valid YouTube URL")
sys.exit(1)
print(f"Video ID: {video_id}")
print("Fetching transcript...")
# Try to get English transcript first
transcript = get_transcript(video_id, ['en', 'en-US', 'en-GB'])
if transcript is None:
# Try to get transcript in any language
print("\nNo English transcript found. Trying to get transcript in original language...")
transcript = get_transcript(video_id, [])
if transcript is None:
print("\nCould not retrieve transcript.")
print("Possible reasons:")
print("- Video does not have captions/subtitles")
print("- Video is live stream")
print("- Video is private")
sys.exit(1)
# Format original transcript
original_formatted = format_transcript(transcript, include_timestamps=True)
# Check if translation is needed (try to detect if it's already English)
original_text = get_transcript_text(transcript)
# Ask user if they want to translate
print("\n" + "=" * 45)
print("Transcript successfully retrieved!")
# Determine if translation is needed
# If transcript is in English, no translation needed
needs_translation = not is_english(original_text)
if needs_translation:
print("\nNote: The transcript appears to be in a non-English language.")
print("Would you like to translate it to English?")
response = input("Enter 'y' for yes or 'n' for no: ").strip().lower()
if response == 'y':
print("\nTranslating to English...")
translated = translate_text(original_text, 'en')
if translated:
# Save translated transcript
output_file = f"transcript_{video_id}_translated.txt"
with open(output_file, 'w', encoding='utf-8') as f:
f.write("TRANSLATED TRANSCRIPT\n")
f.write("=" * 45 + "\n\n")
f.write(translated)
print(f"\nTranslated transcript saved to: {output_file}")
# Save both original and translated
original_file = f"transcript_{video_id}_original.txt"
with open(original_file, 'w', encoding='utf-8') as f:
f.write(original_formatted)
print(f"Original transcript saved to: {original_file}")
# Display original and translated
print("\n" + "=" * 45)
print("ORIGINAL TRANSCRIPT:")
print("=" * 45)
print(original_formatted)
print("\n" + "=" * 45)
print("TRANSLATED TRANSCRIPT:")
print("=" * 45)
print(translated)
else:
print("\nTranslation failed. Saving original transcript...")
output_file = f"transcript_{video_id}.txt"
with open(output_file, 'w', encoding='utf-8') as f:
f.write(original_formatted)
print(f"Transcript saved to: {output_file}")
else:
output_file = f"transcript_{video_id}.txt"
with open(output_file, 'w', encoding='utf-8') as f:
f.write(original_formatted)
print(f"\nTranscript saved to: {output_file}")
else:
# Already in English, just save it
output_file = f"transcript_{video_id}.txt"
with open(output_file, 'w', encoding='utf-8') as f:
f.write(original_formatted)
print(f"\nTranscript saved to: {output_file}")
print("\n" + "=" * 45)
print("TRANSCRIPT:")
print("=" * 45)
print(original_formatted)
def is_english(text):
"""Simple check to determine if text is likely English."""
# Common English words to check against
common_english_words = {
'the', 'be', 'to', 'of', 'and', 'a', 'in', 'that', 'have', 'i',
'it', 'for', 'not', 'on', 'with', 'he', 'as', 'you', 'do', 'at',
'this', 'but', 'his', 'by', 'from', 'they', 'we', 'say', 'her',
'she', 'or', 'an', 'will', 'my', 'one', 'all', 'would', 'there',
'their', 'what', 'so', 'up', 'out', 'if', 'about', 'who', 'get',
'which', 'go', 'me', 'when', 'make', 'can', 'like', 'time', 'no',
'just', 'him', 'know', 'take', 'people', 'into', 'year', 'your',
'good', 'some', 'could', 'them', 'see', 'other', 'than', 'then',
'now', 'look', 'only', 'come', 'its', 'over', 'think', 'also',
'back', 'after', 'use', 'two', 'how', 'our', 'work', 'first', 'well',
'way', 'even', 'new', 'want', 'because', 'any', 'these', 'give', 'day'
}
# Simple heuristic: count common English words
words = set(text.lower().split())
matches = sum(1 for word in words if word in common_english_words)
# If we find many common English words, it's likely English
return matches > len(words) * 0.05 if words else False
if __name__ == "__main__":
main()