All checks were successful
Deploy to Production / Build and Deploy (push) Successful in 2m20s
Translation quality & format preservation: - Word: merge adjacent same-format runs into one unit (sentence-level coherence like inline-tag handling); translate comments/balloons; dedupe textbox collection (was translated twice); RTL no longer overrides center/justify alignment; CJK/Arabic font hints (eastAsia/cs) - PPTX: chart translations now actually reach the output file (ChartPart.blob is read-only — rewrite chart XML in the saved ZIP); CJK typeface hints (a:ea) - Excel: sheet renames no longer break references — rewrite cell formulas (3D/quoted), defined names, data validations, cond. formats - PDF: bold/italic honored (hebo/heit/hebi); table cells never merge; unchanged blocks left untouched (typography preserved, fixes duplicate hyperlinks); attempted/changed stats + route gate now cover PDF; CJK font paths; scanned PDFs via Mistral OCR (detection + admin settings) Features: - formality param (formal/informal) + automatic regional-variant prompts - output_mode=bilingual docx (source above translation) - per-user translation memory on Redis (falls back to LRU), context-hashed - QA report + 0-100 confidence score in job status; L0 on by default - OpenAI-compatible providers: whole chunk in ONE numbered-JSON request (~15x fewer calls) with per-item fallback; base prompt always present (custom prompt no longer replaces translation instructions) Infra & marketing alignment: - plan-based engine gating + vision gating (closes paid-engine leak); /providers/available filtered per plan; 107 languages exposed - zh-CN/zh-TW validation fixed; libmagic disabled on Windows (native crash) - admin: Mistral OCR settings + engine status dashboard; httpx<0.28 pin (TestClient breakage); Prometheus test fixture fixed - marketing docs aligned with code (PDF+OCR, retention, engines, pricing) - security: .env.ionos/.env.production/provider_settings.json removed Tests: 1173 passed / 0 failed (6 network tests deselected: free Google endpoint temporarily blocked from this machine)
138 lines
3.2 KiB
Python
138 lines
3.2 KiB
Python
"""
|
|
Language display names for LLM prompts and UI labels.
|
|
|
|
Single source of truth for code → English name across providers. The map
|
|
covers every code exposed by /api/v1/languages (SUPPORTED_LANGUAGES), so
|
|
prompts say "Translate to Tagalog" instead of "Translate to tl" — LLMs
|
|
translate noticeably better with full language names.
|
|
"""
|
|
|
|
from typing import Dict
|
|
|
|
LANGUAGE_NAMES: Dict[str, str] = {
|
|
"af": "Afrikaans",
|
|
"sq": "Albanian",
|
|
"am": "Amharic",
|
|
"ar": "Arabic",
|
|
"hy": "Armenian",
|
|
"az": "Azerbaijani",
|
|
"eu": "Basque",
|
|
"be": "Belarusian",
|
|
"bn": "Bengali",
|
|
"bs": "Bosnian",
|
|
"bg": "Bulgarian",
|
|
"ca": "Catalan",
|
|
"ceb": "Cebuano",
|
|
"zh": "Chinese",
|
|
"zh-CN": "Chinese (Simplified)",
|
|
"zh-TW": "Chinese (Traditional)",
|
|
"co": "Corsican",
|
|
"hr": "Croatian",
|
|
"cs": "Czech",
|
|
"da": "Danish",
|
|
"nl": "Dutch",
|
|
"en": "English",
|
|
"eo": "Esperanto",
|
|
"et": "Estonian",
|
|
"fi": "Finnish",
|
|
"fr": "French",
|
|
"fy": "Frisian",
|
|
"gl": "Galician",
|
|
"ka": "Georgian",
|
|
"de": "German",
|
|
"el": "Greek",
|
|
"gu": "Gujarati",
|
|
"ht": "Haitian Creole",
|
|
"ha": "Hausa",
|
|
"haw": "Hawaiian",
|
|
"he": "Hebrew",
|
|
"hi": "Hindi",
|
|
"hmn": "Hmong",
|
|
"hu": "Hungarian",
|
|
"is": "Icelandic",
|
|
"ig": "Igbo",
|
|
"id": "Indonesian",
|
|
"ga": "Irish",
|
|
"it": "Italian",
|
|
"ja": "Japanese",
|
|
"jv": "Javanese",
|
|
"kn": "Kannada",
|
|
"kk": "Kazakh",
|
|
"km": "Khmer",
|
|
"rw": "Kinyarwanda",
|
|
"ko": "Korean",
|
|
"ku": "Kurdish",
|
|
"ky": "Kyrgyz",
|
|
"lo": "Lao",
|
|
"la": "Latin",
|
|
"lv": "Latvian",
|
|
"lt": "Lithuanian",
|
|
"lb": "Luxembourgish",
|
|
"mk": "Macedonian",
|
|
"mg": "Malagasy",
|
|
"ms": "Malay",
|
|
"ml": "Malayalam",
|
|
"mt": "Maltese",
|
|
"mi": "Maori",
|
|
"mr": "Marathi",
|
|
"mn": "Mongolian",
|
|
"my": "Myanmar (Burmese)",
|
|
"ne": "Nepali",
|
|
"no": "Norwegian",
|
|
"ny": "Nyanja (Chichewa)",
|
|
"or": "Odia (Oriya)",
|
|
"ps": "Pashto",
|
|
"fa": "Persian (Farsi)",
|
|
"pl": "Polish",
|
|
"pt": "Portuguese",
|
|
"pa": "Punjabi",
|
|
"ro": "Romanian",
|
|
"ru": "Russian",
|
|
"sm": "Samoan",
|
|
"gd": "Scots Gaelic",
|
|
"sr": "Serbian",
|
|
"st": "Sesotho",
|
|
"sn": "Shona",
|
|
"sd": "Sindhi",
|
|
"si": "Sinhala",
|
|
"sk": "Slovak",
|
|
"sl": "Slovenian",
|
|
"so": "Somali",
|
|
"es": "Spanish",
|
|
"su": "Sundanese",
|
|
"sw": "Swahili",
|
|
"sv": "Swedish",
|
|
"tl": "Filipino (Tagalog)",
|
|
"tg": "Tajik",
|
|
"ta": "Tamil",
|
|
"tt": "Tatar",
|
|
"te": "Telugu",
|
|
"th": "Thai",
|
|
"tr": "Turkish",
|
|
"tk": "Turkmen",
|
|
"uk": "Ukrainian",
|
|
"ur": "Urdu",
|
|
"ug": "Uyghur",
|
|
"uz": "Uzbek",
|
|
"vi": "Vietnamese",
|
|
"cy": "Welsh",
|
|
"xh": "Xhosa",
|
|
"yi": "Yiddish",
|
|
"yo": "Yoruba",
|
|
"zu": "Zulu",
|
|
}
|
|
|
|
|
|
def language_name(code: str) -> str:
|
|
"""Full English name for a language code; case-insensitive lookup.
|
|
|
|
Falls back to the code itself for unknown values (and "" for auto/None,
|
|
which callers use to mean "detect the source language").
|
|
"""
|
|
if not code or code == "auto":
|
|
return ""
|
|
name = LANGUAGE_NAMES.get(code)
|
|
if name is None:
|
|
name = LANGUAGE_NAMES.get(code.split("-")[0].lower(), code)
|
|
return name
|