feat(translation): quality pipeline overhaul + new features (audit 2026-08-29)
All checks were successful
Deploy to Production / Build and Deploy (push) Successful in 2m20s

Translation quality & format preservation:
- Word: merge adjacent same-format runs into one unit (sentence-level
  coherence like inline-tag handling); translate comments/balloons;
  dedupe textbox collection (was translated twice); RTL no longer
  overrides center/justify alignment; CJK/Arabic font hints (eastAsia/cs)
- PPTX: chart translations now actually reach the output file
  (ChartPart.blob is read-only — rewrite chart XML in the saved ZIP);
  CJK typeface hints (a:ea)
- Excel: sheet renames no longer break references — rewrite cell
  formulas (3D/quoted), defined names, data validations, cond. formats
- PDF: bold/italic honored (hebo/heit/hebi); table cells never merge;
  unchanged blocks left untouched (typography preserved, fixes duplicate
  hyperlinks); attempted/changed stats + route gate now cover PDF;
  CJK font paths; scanned PDFs via Mistral OCR (detection + admin settings)

Features:
- formality param (formal/informal) + automatic regional-variant prompts
- output_mode=bilingual docx (source above translation)
- per-user translation memory on Redis (falls back to LRU), context-hashed
- QA report + 0-100 confidence score in job status; L0 on by default
- OpenAI-compatible providers: whole chunk in ONE numbered-JSON request
  (~15x fewer calls) with per-item fallback; base prompt always present
  (custom prompt no longer replaces translation instructions)

Infra & marketing alignment:
- plan-based engine gating + vision gating (closes paid-engine leak);
  /providers/available filtered per plan; 107 languages exposed
- zh-CN/zh-TW validation fixed; libmagic disabled on Windows (native crash)
- admin: Mistral OCR settings + engine status dashboard; httpx<0.28 pin
  (TestClient breakage); Prometheus test fixture fixed
- marketing docs aligned with code (PDF+OCR, retention, engines, pricing)
- security: .env.ionos/.env.production/provider_settings.json removed

Tests: 1173 passed / 0 failed (6 network tests deselected: free Google
endpoint temporarily blocked from this machine)
This commit is contained in:
2026-08-29 18:38:09 +02:00
parent 992f13d53c
commit 526c87348f
87 changed files with 6996 additions and 1024 deletions

View File

@@ -38,6 +38,9 @@ class FileCleanupManager:
self.max_file_age_seconds = max_file_age_minutes * 60
self.cleanup_interval = cleanup_interval_minutes * 60
self.max_total_size_bytes = int(max_total_size_gb * 1024 * 1024 * 1024)
# Untracked (orphan) files must be at least this old before deletion:
# protects in-flight files whose Redis metadata is missing or delayed.
self.orphan_grace_seconds = 900
self._running = False
self._task: Optional[asyncio.Task] = None
@@ -191,10 +194,13 @@ class FileCleanupManager:
data = await redis_client.get(key)
if data:
metadata = json.loads(data)
if "file_path" in metadata:
# Normalize path to absolute string for comparison
path_str = str(Path(metadata["file_path"]).absolute())
tracked_paths.add(path_str)
# Metadata uses "input_path" (and possibly other *_path keys);
# collect every path field so tracked files are never
# misclassified as orphans.
for path_key in ("input_path", "file_path", "output_path"):
if path_key in metadata:
path_str = str(Path(metadata[path_key]).absolute())
tracked_paths.add(path_str)
except Exception as e:
logger.warning(f"Failed to fetch tracked paths from Redis: {e}")
redis_available = False
@@ -234,8 +240,11 @@ class FileCleanupManager:
reason = ""
if is_orphan:
should_delete = True
reason = "orphan"
# Never delete a young file as orphan: it may belong to
# a job whose Redis metadata is not yet visible.
if file_age > self.orphan_grace_seconds:
should_delete = True
reason = "orphan"
elif file_age > self.max_file_age_seconds:
should_delete = True
reason = "expired"

View File

@@ -4,7 +4,7 @@ Validates all user inputs before processing
"""
import re
import magic
import sys
import ipaddress
import socket
from pathlib import Path
@@ -13,6 +13,19 @@ from typing import Optional, List, Set, Tuple
from fastapi import UploadFile, HTTPException
import logging
# python-magic shells out to libmagic (a native library). It is known to
# crash with an access violation on some Windows setups — a native crash a
# try/except cannot catch. The ZIP/PDF magic-byte checks below do not need
# libmagic, so it is simply disabled on Windows and treated as optional
# elsewhere (a failed import degrades to the same magic-byte fallback).
if sys.platform == "win32":
magic = None
else:
try:
import magic
except Exception:
magic = None
logger = logging.getLogger(__name__)
@@ -303,13 +316,17 @@ class FileValidator:
def _detect_mime_type(self, content: bytes) -> str:
"""Detect MIME type from file content"""
try:
mime = magic.Magic(mime=True)
return mime.from_buffer(content)
if magic is not None:
mime = magic.Magic(mime=True)
return mime.from_buffer(content)
except Exception:
# Fallback to basic detection
if content.startswith(self.OFFICE_MAGIC_BYTES):
return "application/zip"
return "application/octet-stream"
pass
# Fallback to basic detection (magic bytes, no libmagic needed)
if content.startswith(self.OFFICE_MAGIC_BYTES):
return "application/zip"
if content.startswith(self.PDF_MAGIC_BYTES):
return "application/pdf"
return "application/octet-stream"
def _validate_mime_type(self, mime_type: str, extension: str):
"""Validate MIME type matches extension"""
@@ -446,41 +463,127 @@ class LanguageValidator:
}
LANGUAGE_NAMES = {
"en": "English",
"es": "Spanish",
"fr": "French",
"de": "German",
"it": "Italian",
"pt": "Portuguese",
"ru": "Russian",
"af": "Afrikaans",
"sq": "Albanian",
"am": "Amharic",
"ar": "Arabic",
"hy": "Armenian",
"az": "Azerbaijani",
"eu": "Basque",
"be": "Belarusian",
"bn": "Bengali",
"bs": "Bosnian",
"bg": "Bulgarian",
"ca": "Catalan",
"ceb": "Cebuano",
"zh": "Chinese",
"zh-CN": "Chinese (Simplified)",
"zh-TW": "Chinese (Traditional)",
"ja": "Japanese",
"ko": "Korean",
"ar": "Arabic",
"hi": "Hindi",
"nl": "Dutch",
"pl": "Polish",
"tr": "Turkish",
"sv": "Swedish",
"da": "Danish",
"no": "Norwegian",
"fi": "Finnish",
"co": "Corsican",
"hr": "Croatian",
"cs": "Czech",
"da": "Danish",
"nl": "Dutch",
"en": "English",
"eo": "Esperanto",
"et": "Estonian",
"fi": "Finnish",
"fr": "French",
"fy": "Frisian",
"gl": "Galician",
"ka": "Georgian",
"de": "German",
"el": "Greek",
"th": "Thai",
"vi": "Vietnamese",
"id": "Indonesian",
"uk": "Ukrainian",
"ro": "Romanian",
"gu": "Gujarati",
"ht": "Haitian Creole",
"ha": "Hausa",
"haw": "Hawaiian",
"he": "Hebrew",
"hi": "Hindi",
"hmn": "Hmong",
"hu": "Hungarian",
"is": "Icelandic",
"ig": "Igbo",
"id": "Indonesian",
"ga": "Irish",
"it": "Italian",
"ja": "Japanese",
"jv": "Javanese",
"kn": "Kannada",
"kk": "Kazakh",
"km": "Khmer",
"rw": "Kinyarwanda",
"ko": "Korean",
"ku": "Kurdish",
"ky": "Kyrgyz",
"lo": "Lao",
"la": "Latin",
"lv": "Latvian",
"lt": "Lithuanian",
"lb": "Luxembourgish",
"mk": "Macedonian",
"mg": "Malagasy",
"ms": "Malay",
"ml": "Malayalam",
"mt": "Maltese",
"mi": "Maori",
"mr": "Marathi",
"mn": "Mongolian",
"my": "Myanmar (Burmese)",
"ne": "Nepali",
"no": "Norwegian",
"ny": "Nyanja (Chichewa)",
"or": "Odia (Oriya)",
"ps": "Pashto",
"fa": "Persian (Farsi)",
"pl": "Polish",
"pt": "Portuguese",
"pa": "Punjabi",
"ro": "Romanian",
"ru": "Russian",
"sm": "Samoan",
"gd": "Scots Gaelic",
"sr": "Serbian",
"st": "Sesotho",
"sn": "Shona",
"sd": "Sindhi",
"si": "Sinhala",
"sk": "Slovak",
"sl": "Slovenian",
"so": "Somali",
"es": "Spanish",
"su": "Sundanese",
"sw": "Swahili",
"sv": "Swedish",
"tl": "Filipino (Tagalog)",
"tg": "Tajik",
"ta": "Tamil",
"tt": "Tatar",
"te": "Telugu",
"th": "Thai",
"tr": "Turkish",
"tk": "Turkmen",
"uk": "Ukrainian",
"ur": "Urdu",
"ug": "Uyghur",
"uz": "Uzbek",
"vi": "Vietnamese",
"cy": "Welsh",
"xh": "Xhosa",
"yi": "Yiddish",
"yo": "Yoruba",
"zu": "Zulu",
"auto": "Auto-detect",
}
@classmethod
def validate(cls, language_code: str, field_name: str = "language") -> str:
"""Validate and normalize language code"""
"""Validate and normalize language code.
Matching is case-insensitive so that regional variants stored in
mixed case ("zh-CN", "zh-TW") can be submitted in any case; the
canonical form from SUPPORTED_LANGUAGES is returned.
"""
if not language_code:
raise ValidationError(f"{field_name} est requis", code="missing_language")
@@ -489,9 +592,17 @@ class LanguageValidator:
# Handle common variations
if normalized in ["chinese", "cn"]:
normalized = "zh-CN"
normalized = "zh-cn"
elif normalized in ["chinese-traditional", "tw"]:
normalized = "zh-TW"
normalized = "zh-tw"
# Resolve to the canonical form stored in SUPPORTED_LANGUAGES
# (e.g. "zh-cn" -> "zh-CN"). Unknown codes stay lower-cased and are
# rejected by the membership check below.
for lang in cls.SUPPORTED_LANGUAGES:
if lang.lower() == normalized:
normalized = lang
break
if normalized not in cls.SUPPORTED_LANGUAGES:
raise ValidationError(