feat: centralize common English words for language detection

Move COMMON_ENGLISH_WORDS set from youtube_transcript_downloader.py to language_codes.py as a shared constant. This improves code organization and reusability, allowing other modules to reference the same word list for English language detection
This commit is contained in:
2026-04-07 12:24:52 -04:00
parent 25e9e36372
commit 98b1a676cb
2 changed files with 15 additions and 14 deletions

View File

@@ -24,6 +24,19 @@ COMMON_FALLBACK_LANGUAGES = [
ENGLISH_LANGUAGES = ['en', 'en-US', 'en-GB']
# This can be used as the list of allowed fallback languages
ALLOWED_FALLBACK_LANGUAGES = COMMON_FALLBACK_LANGUAGES
COMMON_ENGLISH_WORDS = {
'the', 'be', 'to', 'of', 'and', 'a', 'in', 'that', 'have', 'i',
'it', 'for', 'not', 'on', 'with', 'he', 'as', 'you', 'do', 'at',
'this', 'but', 'his', 'by', 'from', 'they', 'we', 'say', 'her',
'she', 'or', 'an', 'will', 'my', 'one', 'all', 'would', 'there',
'their', 'what', 'so', 'up', 'out', 'if', 'about', 'who', 'get',
'which', 'go', 'me', 'when', 'make', 'can', 'like', 'time', 'no',
'just', 'him', 'know', 'take', 'people', 'into', 'year', 'your',
'good', 'some', 'could', 'them', 'see', 'other', 'than', 'then',
'now', 'look', 'only', 'come', 'its', 'over', 'think', 'also',
'back', 'after', 'use', 'two', 'how', 'our', 'work', 'first', 'well',
'way', 'even', 'new', 'want', 'because', 'any', 'these', 'give', 'day'
}
if __name__ == "__main__":
print("YouTube Language Codes:")

View File

@@ -191,23 +191,11 @@ def main():
def is_english(text):
"""Simple check to determine if text is likely English."""
# Common English words to check against
common_english_words = {
'the', 'be', 'to', 'of', 'and', 'a', 'in', 'that', 'have', 'i',
'it', 'for', 'not', 'on', 'with', 'he', 'as', 'you', 'do', 'at',
'this', 'but', 'his', 'by', 'from', 'they', 'we', 'say', 'her',
'she', 'or', 'an', 'will', 'my', 'one', 'all', 'would', 'there',
'their', 'what', 'so', 'up', 'out', 'if', 'about', 'who', 'get',
'which', 'go', 'me', 'when', 'make', 'can', 'like', 'time', 'no',
'just', 'him', 'know', 'take', 'people', 'into', 'year', 'your',
'good', 'some', 'could', 'them', 'see', 'other', 'than', 'then',
'now', 'look', 'only', 'come', 'its', 'over', 'think', 'also',
'back', 'after', 'use', 'two', 'how', 'our', 'work', 'first', 'well',
'way', 'even', 'new', 'want', 'because', 'any', 'these', 'give', 'day'
}
# Simple heuristic: count common English words
words = set(text.lower().split())
matches = sum(1 for word in words if word in common_english_words)
matches = sum(1 for word in words if word in language_codes.COMMON_ENGLISH_WORDS)
# If we find many common English words, it's likely English
return matches > len(words) * 0.05 if words else False