import pycld2 as cld2 from utils.logger import session_logger import re import unicodedata import asyncio import concurrent.futures class LanguageDetector: LANGUAGE_UNICODE_RANGES = { 'zh': [ # 中文 (0x4E00, 0x9FFF), # CJK统一汉字 (0x3400, 0x4DBF), # CJK扩展A (0xF900, 0xFAFF), # CJK兼容汉字 ] } LANGUAGE_FEATURES = { 'zh': ['的', '了', '是', '我', '你', '他', '们', '在', '有', '和', '这', '那', '什么', '怎么', '为什么', '好的', '嗯', '啊', '哦', '吧', '呢', '吗', '哈', '嘿', '喂', '您好', '谢谢', '对不起', '没关系', '好啊', '行', '可以', '不行', '不可以', '是的', '不是', '要', '不要', '能', '能不能', '请问', '请', '帮我', '告诉我', '知道', '不知道', '明白', '不明白', '对', '不对', '听见', '听不见'], 'en': ['i', 'you', 'he', 'she', 'we', 'they', 'this', 'that', 'what', 'how', 'why', 'where', 'when', 'okay', 'yes', 'no', 'yeah', 'oh', 'hey', 'hi', 'hello', 'thanks', 'thank', 'sorry', 'please', 'can', 'could', 'would', 'will', 'want', 'need', 'like', 'tell', 'know', 'think', 'mean', 'say', 'said', 'help', 'hear', 'listen', 'understand', 'right', 'wrong', 'good', 'bad', 'ok', 'alright', 'excuse', 'pardon', 'well', 'sure', 'maybe', 'perhaps'], 'de': ['ich', 'du', 'er', 'sie', 'es', 'wir', 'ihr', 'sie', 'das', 'was', 'wie', 'warum', 'wo', 'wann', 'ja', 'nein', 'okay', 'hallo', 'danke', 'bitte', 'entschuldigung', 'tschüss', 'guten', 'tag', 'morgen', 'abend', 'kann', 'könnte', 'würde', 'möchte', 'wollen', 'brauchen', 'verstehe', 'weiß', 'denke', 'meine', 'helfen', 'hören', 'zuhören', 'richtig', 'falsch', 'gut', 'schlecht', 'alles', 'klar', 'natürlich', 'vielleicht'], 'fr': ['je', 'tu', 'il', 'elle', 'nous', 'vous', 'ils', 'elles', 'ce', 'ça', 'quoi', 'comment', 'pourquoi', 'où', 'quand', 'oui', 'non', 'salut', 'bonjour', 'merci', 'pardon', 'excusez', 'au revoir', 'peux', 'pouvez', 'voulez', 'veux', 'besoin', 'comprends', 'sais', 'pense', 'dire', 'aider', 'entendre', 'écouter', 'correct', 'incorrect', 'bon', 'bien', 'mauvais', 'mal', 'd\'accord', 'bien sûr', 'peut-être'], 'es': ['yo', 'tú', 'él', 'ella', 'nosotros', 'vosotros', 'ellos', 'ellas', 'esto', 'eso', 'qué', 'cómo', 'por qué', 'dónde', 'cuándo', 'sí', 'no', 'hola', 'gracias', 'perdón', 'disculpe', 'adiós', 'puede', 'puedes', 'quiere', 'quiero', 'necesito', 'entiendo', 'sé', 'pienso', 'digo', 'ayudar', 'oír', 'escuchar', 'correcto', 'incorrecto', 'bueno', 'malo', 'vale', 'claro', 'quizás', 'tal vez'] } LANGUAGE_PUNCTUATION = { 'zh': ['。', ',', '、', ':', ';', '!', '?', '"', '"', '(', ')', '《', '》', '【', '】', '——', '……', '~'], 'en': [], # 英语标点大多通用,不作为特征 'de': ['„', '"'], # 德语特有引号 'fr': ['«', '»', '…'], # 法语特有引号和省略号 'es': ['¿', '¡'] # 西班牙语特有的倒置问号和感叹号 } SPEECH_FILLERS = { 'zh': ['嗯', '呃', '啊', '哦', '唉', '那个', '这个', '就是', '其实', '然后', '所以', '但是'], 'en': ['um', 'uh', 'er', 'ah', 'like', 'you know', 'i mean', 'well', 'so', 'anyway', 'actually'], 'de': ['äh', 'ähm', 'hmm', 'na ja', 'also', 'sozusagen', 'quasi', 'naja', 'eigentlich', 'tja'], 'fr': ['euh', 'ben', 'bah', 'eh bien', 'bon', 'alors', 'donc', 'voilà', 'en fait', 'quoi'], 'es': ['eh', 'este', 'em', 'pues', 'bueno', 'o sea', 'vale', 'vamos', 'mira', 'entonces'] } SPEECH_PATTERNS = { 'zh': [r'^(你好|您好|喂|在吗|请问|那个|这个|帮我|麻烦|请|要|需要)'], 'en': [r'^(hello|hi|hey|excuse me|sorry|please|can you|could you|would you|I want|I need)'], 'de': [r'^(hallo|guten tag|entschuldigung|bitte|können sie|könntest du|ich möchte|ich brauche)'], 'fr': [r'^(bonjour|salut|excusez-moi|pardon|s\'il vous plaît|pouvez-vous|pourriez-vous|je voudrais|j\'ai besoin)'], 'es': [r'^(hola|buenos días|perdón|disculpe|por favor|puede usted|puedes|quiero|necesito)'] } _executor = concurrent.futures.ThreadPoolExecutor( max_workers=4, thread_name_prefix="lang_detector" ) @staticmethod async def detect_language(text): try: SUPPORTED_LANGUAGES = ["zh", "en", "fr", "de", "es"] if not text: return "en" # 默认返回英文 cleaned_text = ' '.join(text.lower().split()) loop = asyncio.get_running_loop() speech_pattern_lang = await loop.run_in_executor( LanguageDetector._executor, LanguageDetector._detect_by_speech_patterns, cleaned_text ) if speech_pattern_lang != "unknown": return speech_pattern_lang char_based_lang = await loop.run_in_executor( LanguageDetector._executor, LanguageDetector._detect_by_characters, cleaned_text ) if char_based_lang != "unknown": return char_based_lang feature_based_lang = await loop.run_in_executor( LanguageDetector._executor, LanguageDetector._detect_by_features, cleaned_text ) if feature_based_lang != "unknown": return feature_based_lang try: is_reliable, _, details = cld2.detect(cleaned_text) if is_reliable and details[0][1] in SUPPORTED_LANGUAGES: return details[0][1] except: pass for ch in cleaned_text: if '\u4e00' <= ch <= '\u9fff': # 检测中文字符 return "zh" return "en" # 默认返回英语 except Exception as e: session_logger.error("system", "language_detector", f"语言检测错误: {str(e)}") return "en" # 出错时默认使用英文 @staticmethod def _detect_by_speech_patterns(text: str) -> str: """检测常见的语音对话开头模式""" for lang, patterns in LanguageDetector.SPEECH_PATTERNS.items(): for pattern in patterns: if re.search(pattern, text): return lang return "unknown" @staticmethod def _detect_by_characters(text: str) -> str: """基于字符分布的语言检测,特别适合短文本""" text = text.strip() text = text.lower() text_len = max(1, len(text)) char_counts = {lang: 0 for lang in LanguageDetector.LANGUAGE_UNICODE_RANGES} for char in text: code_point = ord(char) for lang, ranges in LanguageDetector.LANGUAGE_UNICODE_RANGES.items(): for start, end in ranges: if start <= code_point <= end: char_counts[lang] += 1 for lang, count in char_counts.items(): ratio = count / text_len if lang == 'zh' and ratio > 0.12: return 'zh' latin_chars = sum(1 for c in text if unicodedata.category(c).startswith('L')) latin_ratio = latin_chars / text_len if latin_ratio > 0.4: # 降低拉丁字符检测阈值,更灵敏地捕捉语音转文本 if any(c in text for c in "äöüß"): return "de" # 德语特有字符 elif any(c in text for c in "éèêëàâçùûüÿôœæ"): return "fr" # 法语特有字符 elif any(c in text for c in "áéíóúñ¿¡"): return "es" # 西班牙语特有字符 if "¿" in text or "¡" in text: return "es" # 西班牙语特有标点 elif "«" in text or "»" in text: return "fr" # 法语引号 return "en" # 默认英语 return "unknown" @staticmethod def _detect_by_features(text: str) -> str: """基于特征词和语气词的语言检测,针对口语场景优化""" text_lower = text.lower() matches = {lang: 0 for lang in LanguageDetector.LANGUAGE_FEATURES} for lang, features in LanguageDetector.LANGUAGE_FEATURES.items(): for word in features: if lang == 'zh': # 中文不需要空格分隔 if word in text_lower: matches[lang] += 2 elif f" {word} " in f" {text_lower} ": matches[lang] += 1 for lang, fillers in LanguageDetector.SPEECH_FILLERS.items(): for filler in fillers: if lang == 'zh': count = text_lower.count(filler) else: count = text_lower.count(f" {filler} ") matches[lang] += count * 3 for lang, puncts in LanguageDetector.LANGUAGE_PUNCTUATION.items(): for punct in puncts: matches[lang] += text.count(punct) * 5 nonzero_matches = {lang: score for lang, score in matches.items() if score > 0} if nonzero_matches: best_lang = max(nonzero_matches.items(), key=lambda x: x[1]) if best_lang[1] > 0: return best_lang[0] return "unknown"