feat: centralize common English words for language detection
Move COMMON_ENGLISH_WORDS set from youtube_transcript_downloader.py to language_codes.py as a shared constant. This improves code organization and reusability, allowing other modules to reference the same word list for English language detection
This commit is contained in:
@@ -24,6 +24,19 @@ COMMON_FALLBACK_LANGUAGES = [
|
||||
ENGLISH_LANGUAGES = ['en', 'en-US', 'en-GB']
|
||||
# This can be used as the list of allowed fallback languages
|
||||
ALLOWED_FALLBACK_LANGUAGES = COMMON_FALLBACK_LANGUAGES
|
||||
COMMON_ENGLISH_WORDS = {
|
||||
'the', 'be', 'to', 'of', 'and', 'a', 'in', 'that', 'have', 'i',
|
||||
'it', 'for', 'not', 'on', 'with', 'he', 'as', 'you', 'do', 'at',
|
||||
'this', 'but', 'his', 'by', 'from', 'they', 'we', 'say', 'her',
|
||||
'she', 'or', 'an', 'will', 'my', 'one', 'all', 'would', 'there',
|
||||
'their', 'what', 'so', 'up', 'out', 'if', 'about', 'who', 'get',
|
||||
'which', 'go', 'me', 'when', 'make', 'can', 'like', 'time', 'no',
|
||||
'just', 'him', 'know', 'take', 'people', 'into', 'year', 'your',
|
||||
'good', 'some', 'could', 'them', 'see', 'other', 'than', 'then',
|
||||
'now', 'look', 'only', 'come', 'its', 'over', 'think', 'also',
|
||||
'back', 'after', 'use', 'two', 'how', 'our', 'work', 'first', 'well',
|
||||
'way', 'even', 'new', 'want', 'because', 'any', 'these', 'give', 'day'
|
||||
}
|
||||
|
||||
if __name__ == "__main__":
|
||||
print("YouTube Language Codes:")
|
||||
|
||||
@@ -191,23 +191,11 @@ def main():
|
||||
def is_english(text):
|
||||
"""Simple check to determine if text is likely English."""
|
||||
# Common English words to check against
|
||||
common_english_words = {
|
||||
'the', 'be', 'to', 'of', 'and', 'a', 'in', 'that', 'have', 'i',
|
||||
'it', 'for', 'not', 'on', 'with', 'he', 'as', 'you', 'do', 'at',
|
||||
'this', 'but', 'his', 'by', 'from', 'they', 'we', 'say', 'her',
|
||||
'she', 'or', 'an', 'will', 'my', 'one', 'all', 'would', 'there',
|
||||
'their', 'what', 'so', 'up', 'out', 'if', 'about', 'who', 'get',
|
||||
'which', 'go', 'me', 'when', 'make', 'can', 'like', 'time', 'no',
|
||||
'just', 'him', 'know', 'take', 'people', 'into', 'year', 'your',
|
||||
'good', 'some', 'could', 'them', 'see', 'other', 'than', 'then',
|
||||
'now', 'look', 'only', 'come', 'its', 'over', 'think', 'also',
|
||||
'back', 'after', 'use', 'two', 'how', 'our', 'work', 'first', 'well',
|
||||
'way', 'even', 'new', 'want', 'because', 'any', 'these', 'give', 'day'
|
||||
}
|
||||
|
||||
|
||||
# Simple heuristic: count common English words
|
||||
words = set(text.lower().split())
|
||||
matches = sum(1 for word in words if word in common_english_words)
|
||||
matches = sum(1 for word in words if word in language_codes.COMMON_ENGLISH_WORDS)
|
||||
|
||||
# If we find many common English words, it's likely English
|
||||
return matches > len(words) * 0.05 if words else False
|
||||
|
||||
Reference in New Issue
Block a user