AnimaStudio / i18n.py
lulavc
fix: move theme/css from gr.Blocks() to demo.launch() for Gradio 6.x compatibility
43f8b96
Raw History Blame Contribute Delete
18.7 kB
"""Internationalization: TTS language list, example texts, and UI translations."""
MAX_TEXT_LEN = 500
MAX_AUDIO_SEC = 30
TTS_LANGUAGES = [
"Arabic", "Danish", "German", "Greek", "English",
"Spanish", "Finnish", "French", "Hebrew", "Hindi",
"Italian", "Japanese", "Korean", "Malay", "Dutch",
"Norwegian", "Polish", "Portuguese", "Russian", "Swedish",
"Swahili", "Turkish", "Chinese",
]
# ── Examples per UI language ──────────────────────────────────────────────────
# Format: [text, tts_language, emotion]
EXAMPLES = {
"🇺🇸 English": [
["Hello! Welcome to this presentation. Today I'll be sharing some exciting insights about artificial intelligence and how it's changing the world around us.", "English", 0.5],
["I'm thrilled to announce the launch of our new project. After months of hard work and dedication, we've created something truly special I can't wait to share with you.", "English", 0.7],
["Good morning, students! Today's lecture covers neural networks — the backbone of modern AI. By the end of this session you'll understand how machines learn from data.", "English", 0.4],
["Breaking news: scientists have discovered a new method for sustainable energy production that could revolutionise how we power our cities. More details coming up.", "English", 0.6],
],
"🇧🇷 Português": [
["Olá a todos! Sejam bem-vindos a esta apresentação. Hoje vou compartilhar com vocês algumas descobertas incríveis sobre inteligência artificial e como ela está transformando o nosso mundo.", "Portuguese", 0.5],
["Estou muito animado para anunciar o lançamento do nosso novo projeto. Depois de meses de trabalho dedicado, criamos algo verdadeiramente especial que mal posso esperar para mostrar a vocês.", "Portuguese", 0.7],
["Bom dia, estudantes! A aula de hoje aborda redes neurais — a base da IA moderna. Ao final desta sessão, vocês vão entender como as máquinas aprendem com os dados.", "Portuguese", 0.4],
["Esta receita tradicional foi passada de geração em geração na minha família. Hoje vou ensinar como preparar um prato delicioso que vai impressionar todos os seus convidados.", "Portuguese", 0.6],
],
"🇪🇸 Español": [
["¡Hola a todos! Bienvenidos a esta presentación. Hoy voy a compartir con ustedes algunos descubrimientos fascinantes sobre la inteligencia artificial y cómo está transformando nuestro mundo.", "Spanish", 0.5],
["Estoy muy emocionado de anunciar el lanzamiento de nuestro nuevo proyecto. Después de meses de arduo trabajo hemos creado algo verdaderamente especial que no puedo esperar para mostrarles.", "Spanish", 0.7],
["Buenos días, estudiantes. La clase de hoy trata sobre las redes neuronales, la columna vertebral de la IA moderna. Al final de esta sesión entenderán cómo las máquinas aprenden de los datos.", "Spanish", 0.4],
["Esta receta tradicional ha pasado de generación en generación en mi familia. Hoy les enseñaré cómo preparar un plato delicioso que impresionará a todos sus invitados.", "Spanish", 0.6],
],
"🇪🇬 عربي": [
["مرحباً بالجميع! أهلاً وسهلاً بكم في هذا العرض التقديمي. اليوم سأشارككم بعض الاكتشافات المثيرة حول الذكاء الاصطناعي وكيف يُغيّر عالمنا من حولنا.", "Arabic", 0.5],
["يسعدني الإعلان عن إطلاق مشروعنا الجديد. بعد أشهر من العمل الدؤوب أبدعنا شيئاً رائعاً حقاً لا أصبر على مشاركته معكم جميعاً.", "Arabic", 0.7],
["صباح الخير أيها الطلاب! محاضرة اليوم تتناول الشبكات العصبية التي تمثّل الأساس التقني للذكاء الاصطناعي الحديث. بنهاية هذه الجلسة ستفهمون كيف تتعلم الآلات من البيانات.", "Arabic", 0.4],
["هذه الوصفة التقليدية انتقلت من جيل إلى جيل في عائلتي. اليوم سأعلّمكم كيفية تحضير طبق شهي سيُبهر جميع ضيوفكم ويجعلهم يطلبون المزيد.", "Arabic", 0.6],
],
}
ALL_EXAMPLES_FLAT = [ex for exs in EXAMPLES.values() for ex in exs]
# ── UI translations ────────────────────────────────────────────────────────────
T: dict[str, dict[str, str]] = {
"🇺🇸 English": {
# Phase 1: Create Video
"tab_create": "🎬 Create Video",
"tagline": "AI Talking Head Video Creator",
"input_mode_label": "Audio Input",
"mode_text": "Text to Speech",
"mode_audio": "Upload Audio",
"portrait_label": "Portrait Photo",
"portrait_info": "Upload a clear, front-facing face photo",
"text_label": "Text",
"text_ph": "Type what you want the avatar to say...",
"tts_lang_label": "Speech Language",
"voice_ref_label": "Voice Reference",
"voice_ref_info": "Optional: upload audio to clone the voice style",
"emotion_label": "Emotion Intensity",
"emotion_info": "0 = neutral · 1 = very expressive",
"audio_label": "Audio File",
"audio_info": "Upload WAV, MP3, or FLAC · max 30 seconds",
"aspect_label": "Format",
"advanced": "⚙️ Advanced Settings",
"steps_label": "Inference Steps",
"steps_info": "More steps = higher quality, slower",
"guidance_label": "Guidance Scale",
"guidance_info": "Higher = follows audio more strictly",
"generate": "🎬 Generate Video",
"output_label": "Generated Video",
"examples_header": "### 💡 Try These Examples",
"err_no_portrait": "Please upload a portrait photo.",
"err_no_text": "Please enter some text.",
"err_no_audio": "Please upload an audio file.",
"err_text_long": f"Text too long (max {MAX_TEXT_LEN} characters).",
"err_audio_long": f"Audio too long (max {MAX_AUDIO_SEC} seconds).",
"err_oom": "GPU out of memory. Try a smaller format or fewer steps.",
"err_no_face": "No face detected. Please upload a clear front-facing portrait.",
"err_model": "Model not loaded. Please refresh and try again.",
# Phase 2: Dub Video
"tab_dub": "🎙️ Dub Video",
"dub_tagline": "Dub any video into 23 languages",
"dub_video_label": "Input Video",
"dub_video_info": "Upload a video to dub (max 60 seconds)",
"dub_target_label": "Target Language",
"dub_voice_label": "Voice Reference",
"dub_voice_info": "Optional: upload audio to clone voice style for dubbing",
"dub_emotion_label": "Emotion Intensity",
"dub_btn": "🎙️ Dub Video",
"dub_output_label": "Dubbed Video",
"dub_transcript": "Detected Transcript",
"dub_translation": "Translation",
"dub_status": "Status",
"dub_details": "Details",
"err_no_video": "Please upload a video.",
"err_video_long": "Video too long (max 60 seconds).",
"err_translate": "Translation failed. Please try again.",
"err_transcribe": "Transcription failed. Please try again.",
"err_dub_text_long": "Transcription too long to synthesize. Please use a shorter video.",
},
"🇧🇷 Português": {
# Phase 1: Create Video
"tab_create": "🎬 Criar Vídeo",
"tagline": "Criador de Vídeo Avatar com IA",
"input_mode_label": "Entrada de Áudio",
"mode_text": "Texto para Fala",
"mode_audio": "Enviar Áudio",
"portrait_label": "Foto Retrato",
"portrait_info": "Envie uma foto frontal clara do rosto",
"text_label": "Texto",
"text_ph": "Digite o que você quer que o avatar diga...",
"tts_lang_label": "Idioma da Fala",
"voice_ref_label": "Referência de Voz",
"voice_ref_info": "Opcional: envie um áudio para clonar o estilo de voz",
"emotion_label": "Intensidade da Emoção",
"emotion_info": "0 = neutro · 1 = muito expressivo",
"audio_label": "Arquivo de Áudio",
"audio_info": "Envie WAV, MP3 ou FLAC · máx. 30 segundos",
"aspect_label": "Formato",
"advanced": "⚙️ Configurações Avançadas",
"steps_label": "Etapas de Inferência",
"steps_info": "Mais etapas = maior qualidade, mais lento",
"guidance_label": "Escala de Orientação",
"guidance_info": "Maior = segue o áudio com mais precisão",
"generate": "🎬 Gerar Vídeo",
"output_label": "Vídeo Gerado",
"examples_header": "### 💡 Experimente Estes Exemplos",
"err_no_portrait": "Por favor, envie uma foto retrato.",
"err_no_text": "Por favor, insira algum texto.",
"err_no_audio": "Por favor, envie um arquivo de áudio.",
"err_text_long": f"Texto muito longo (máx. {MAX_TEXT_LEN} caracteres).",
"err_audio_long": f"Áudio muito longo (máx. {MAX_AUDIO_SEC} segundos).",
"err_oom": "GPU sem memória. Tente um formato menor ou menos etapas.",
"err_no_face": "Nenhum rosto detectado. Envie uma foto retrato frontal clara.",
"err_model": "Modelo não carregado. Atualize a página e tente novamente.",
# Phase 2: Dub Video
"tab_dub": "🎙️ Dublar Vídeo",
"dub_tagline": "Duble qualquer vídeo em 23 idiomas",
"dub_video_label": "Vídeo de Entrada",
"dub_video_info": "Envie um vídeo para dublar (máx. 60 segundos)",
"dub_target_label": "Idioma de Destino",
"dub_voice_label": "Referência de Voz",
"dub_voice_info": "Opcional: envie áudio para clonar o estilo de voz na dublagem",
"dub_emotion_label": "Intensidade da Emoção",
"dub_btn": "🎙️ Dublar Vídeo",
"dub_output_label": "Vídeo Dublado",
"dub_transcript": "Transcrição Detectada",
"dub_translation": "Tradução",
"dub_status": "Status",
"dub_details": "Detalhes",
"err_no_video": "Por favor, envie um vídeo.",
"err_video_long": "Vídeo muito longo (máx. 60 segundos).",
"err_translate": "Tradução falhou. Por favor, tente novamente.",
"err_transcribe": "Transcrição falhou. Por favor, tente novamente.",
"err_dub_text_long": "Transcrição longa demais para sintetizar. Use um vídeo mais curto.",
},
"🇪🇸 Español": {
# Phase 1: Create Video
"tab_create": "🎬 Crear Vídeo",
"tagline": "Creador de Vídeo Avatar con IA",
"input_mode_label": "Entrada de Audio",
"mode_text": "Texto a Voz",
"mode_audio": "Subir Audio",
"portrait_label": "Foto Retrato",
"portrait_info": "Sube una foto frontal clara del rostro",
"text_label": "Texto",
"text_ph": "Escribe lo que quieres que diga el avatar...",
"tts_lang_label": "Idioma del Habla",
"voice_ref_label": "Referencia de Voz",
"voice_ref_info": "Opcional: sube un audio para clonar el estilo de voz",
"emotion_label": "Intensidad Emocional",
"emotion_info": "0 = neutro · 1 = muy expresivo",
"audio_label": "Archivo de Audio",
"audio_info": "Sube WAV, MP3 o FLAC · máx. 30 segundos",
"aspect_label": "Formato",
"advanced": "⚙️ Configuración Avanzada",
"steps_label": "Pasos de Inferencia",
"steps_info": "Más pasos = mayor calidad, más lento",
"guidance_label": "Escala de Guía",
"guidance_info": "Mayor = sigue el audio con más precisión",
"generate": "🎬 Generar Vídeo",
"output_label": "Vídeo Generado",
"examples_header": "### 💡 Prueba Estos Ejemplos",
"err_no_portrait": "Por favor, sube una foto retrato.",
"err_no_text": "Por favor, ingresa algún texto.",
"err_no_audio": "Por favor, sube un archivo de audio.",
"err_text_long": f"Texto demasiado largo (máx. {MAX_TEXT_LEN} caracteres).",
"err_audio_long": f"Audio demasiado largo (máx. {MAX_AUDIO_SEC} segundos).",
"err_oom": "Sin memoria GPU. Prueba un formato menor o menos pasos.",
"err_no_face": "No se detectó rostro. Sube una foto retrato frontal clara.",
"err_model": "Modelo no cargado. Recarga la página e intenta de nuevo.",
# Phase 2: Dub Video
"tab_dub": "🎙️ Doblar Vídeo",
"dub_tagline": "Dobla cualquier vídeo a 23 idiomas",
"dub_video_label": "Vídeo de Entrada",
"dub_video_info": "Sube un vídeo para doblar (máx. 60 segundos)",
"dub_target_label": "Idioma de Destino",
"dub_voice_label": "Referencia de Voz",
"dub_voice_info": "Opcional: sube audio para clonar el estilo de voz en el doblaje",
"dub_emotion_label": "Intensidad Emocional",
"dub_btn": "🎙️ Doblar Vídeo",
"dub_output_label": "Vídeo Doblado",
"dub_transcript": "Transcripción Detectada",
"dub_translation": "Traducción",
"dub_status": "Estado",
"dub_details": "Detalles",
"err_no_video": "Por favor, sube un vídeo.",
"err_video_long": "Vídeo demasiado largo (máx. 60 segundos).",
"err_translate": "Traducción fallida. Por favor, inténtalo de nuevo.",
"err_transcribe": "Transcripción fallida. Por favor, inténtalo de nuevo.",
"err_dub_text_long": "Transcripción demasiado larga. Usa un vídeo más corto.",
},
"🇪🇬 عربي": {
# Phase 1: Create Video
"tab_create": "🎬 إنشاء فيديو",
"tagline": "منشئ فيديو الأفاتار بالذكاء الاصطناعي",
"input_mode_label": "مدخل الصوت",
"mode_text": "نص إلى كلام",
"mode_audio": "رفع ملف صوتي",
"portrait_label": "صورة الوجه",
"portrait_info": "ارفع صورة واضحة للوجه من الأمام",
"text_label": "النص",
"text_ph": "اكتب ما تريد أن يقوله الأفاتار...",
"tts_lang_label": "لغة الكلام",
"voice_ref_label": "مرجع الصوت",
"voice_ref_info": "اختياري: ارفع ملفاً صوتياً لاستنساخ أسلوب الصوت",
"emotion_label": "شدة التعبير العاطفي",
"emotion_info": "0 = محايد · 1 = تعبيري جداً",
"audio_label": "الملف الصوتي",
"audio_info": "ارفع WAV أو MP3 أو FLAC · الحد الأقصى 30 ثانية",
"aspect_label": "التنسيق",
"advanced": "⚙️ الإعدادات المتقدمة",
"steps_label": "خطوات الاستدلال",
"steps_info": "المزيد من الخطوات = جودة أعلى، وقت أطول",
"guidance_label": "مقياس التوجيه",
"guidance_info": "أعلى = يتبع الصوت بدقة أكبر",
"generate": "🎬 توليد الفيديو",
"output_label": "الفيديو المُنشأ",
"examples_header": "### 💡 جرّب هذه الأمثلة",
"err_no_portrait": "الرجاء رفع صورة وجه.",
"err_no_text": "الرجاء إدخال نص.",
"err_no_audio": "الرجاء رفع ملف صوتي.",
"err_text_long": f"النص طويل جداً (الحد الأقصى {MAX_TEXT_LEN} حرف).",
"err_audio_long": f"الصوت طويل جداً (الحد الأقصى {MAX_AUDIO_SEC} ثانية).",
"err_oom": "نفدت ذاكرة GPU. جرّب تنسيقاً أصغر أو خطوات أقل.",
"err_no_face": "لم يُكتشف أي وجه. ارفع صورة وجه واضحة من الأمام.",
"err_model": "النموذج غير محمّل. أعد تحميل الصفحة وحاول مجدداً.",
# Phase 2: Dub Video
"tab_dub": "🎙️ دبلجة فيديو",
"dub_tagline": "دبلج أي فيديو إلى 23 لغة",
"dub_video_label": "الفيديو المُدخل",
"dub_video_info": "ارفع فيديو للدبلجة (الحد الأقصى 60 ثانية)",
"dub_target_label": "اللغة الهدف",
"dub_voice_label": "مرجع الصوت",
"dub_voice_info": "اختياري: ارفع ملفاً صوتياً لاستنساخ أسلوب الصوت في الدبلجة",
"dub_emotion_label": "شدة التعبير العاطفي",
"dub_btn": "🎙️ دبلجة الفيديو",
"dub_output_label": "الفيديو المدبلج",
"dub_transcript": "النص المُكتشف",
"dub_translation": "الترجمة",
"dub_status": "الحالة",
"dub_details": "التفاصيل",
"err_no_video": "الرجاء رفع فيديو.",
"err_video_long": "الفيديو طويل جداً (الحد الأقصى 60 ثانية).",
"err_translate": "فشلت الترجمة. الرجاء المحاولة مجدداً.",
"err_transcribe": "فشل النسخ. الرجاء المحاولة مجدداً.",
"err_dub_text_long": "النص المُكتشف طويل جداً. استخدم مقطعاً أقصر.",
},
}