Some checks failed
Deploy to Production / Build and Deploy (push) Failing after 2m14s
Foundations:
- TranslationSegment model + migration f7e8d9c0b1a2 (segments, workspaces,
workspace_members, glossaries.workspace_id)
- SegmentRecorder injected into all 4 translators: unique (source,
translation) pairs captured per job and persisted (best-effort)
- set_segment_overrides: human-reviewed translations applied verbatim on
rebuild — top priority over TM and provider, zero API calls
Review API (routes/review_routes.py):
- GET /translations/{id}/segments (owner or job token)
- PATCH /segments/{id} edit/approve — feeds the per-user TM so approved
translations are reused in later jobs
- POST /translations/{id}/rebuild — rebuild document with reviewed text
- GET/POST /translations/{id}/xliff — XLIFF 1.2 export/import (edited
segments export their reviewed text)
Review editor (frontend /dashboard/reviews/[jobId]):
- side-by-side source/translation table, inline edit, approve (single or
all), rebuild & download (auth blob), XLIFF export/import, 13 locales
- 'Relire et corriger' link on the translation-complete screen
Team workspaces (routes/workspace_routes.py + /dashboard/teams):
- Workspace/WorkspaceMember models, roles owner/admin/member
- create (Business plan), list with seat usage, invite by email with
seat-limit enforcement (Business=5, Enterprise unlimited), removal
- shared glossaries: workspace members can use a glossary shared to their
workspace (access check extended)
Tests: 1184 passed / 0 failed (11 new: recorder, overrides, docx
capture->rebuild e2e, XLIFF structure/escaping, seats, workspace CRUD,
shared glossary access)
276 lines
9.4 KiB
Python
276 lines
9.4 KiB
Python
"""
|
|
Glossary Service for Translation
|
|
Story 3.10: Glossaires - Application lors Traduction LLM
|
|
|
|
Provides functions to retrieve glossary terms and format them for LLM prompts.
|
|
"""
|
|
|
|
import logging
|
|
from typing import List, Dict, Any, Optional
|
|
|
|
from database.connection import get_sync_session
|
|
from database.models import Glossary, GlossaryTerm
|
|
from utils.exceptions import GlossaryNotFoundError
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def _user_workspace_ids(session, user_id: str) -> List[str]:
|
|
"""Workspaces the user belongs to (for shared-glossary access)."""
|
|
from database.models import WorkspaceMember
|
|
|
|
rows = (
|
|
session.query(WorkspaceMember.workspace_id)
|
|
.filter(WorkspaceMember.user_id == user_id)
|
|
.all()
|
|
)
|
|
return [r[0] for r in rows]
|
|
|
|
|
|
def _glossary_accessible(session, glossary: Glossary, user_id: str) -> bool:
|
|
"""Owner always; otherwise any workspace the glossary is shared with."""
|
|
if str(glossary.user_id) == str(user_id):
|
|
return True
|
|
if glossary.workspace_id:
|
|
return glossary.workspace_id in _user_workspace_ids(session, user_id)
|
|
return False
|
|
|
|
|
|
def get_glossary_terms(glossary_id: str, user_id: str) -> Dict[str, Any]:
|
|
"""
|
|
Retrieve glossary terms and metadata for a glossary the user can access
|
|
(owner, or member of the workspace the glossary is shared with).
|
|
|
|
Args:
|
|
glossary_id: UUID of the glossary
|
|
user_id: UUID of the user
|
|
|
|
Returns:
|
|
Dict with 'source_language' and 'terms' (list of dicts with source, target, translations)
|
|
|
|
Raises:
|
|
GlossaryNotFoundError: If glossary doesn't exist or isn't accessible
|
|
"""
|
|
try:
|
|
with get_sync_session() as session:
|
|
glossary = (
|
|
session.query(Glossary).filter(Glossary.id == glossary_id).first()
|
|
)
|
|
|
|
if not glossary or not _glossary_accessible(session, glossary, user_id):
|
|
raise GlossaryNotFoundError(
|
|
message="Glossaire introuvable ou vous n'avez pas accès à cette ressource.",
|
|
details={"glossary_id": glossary_id}
|
|
)
|
|
|
|
terms = (
|
|
session.query(GlossaryTerm)
|
|
.filter(GlossaryTerm.glossary_id == glossary_id)
|
|
.all()
|
|
)
|
|
|
|
result = [{
|
|
"source": term.source,
|
|
"target": term.target,
|
|
"translations": term.translations or {}
|
|
} for term in terms]
|
|
|
|
logger.info(
|
|
f"Retrieved {len(result)} terms from glossary {glossary_id} for user {user_id}"
|
|
)
|
|
|
|
return {
|
|
"source_language": glossary.source_language or "fr",
|
|
"target_language": getattr(glossary, "target_language", None) or "multi",
|
|
"terms": result,
|
|
}
|
|
|
|
except GlossaryNotFoundError:
|
|
raise
|
|
except Exception as e:
|
|
logger.error(f"Error retrieving glossary {glossary_id}: {e}")
|
|
raise GlossaryNotFoundError(
|
|
message="Erreur lors de la récupération du glossaire.",
|
|
details={"glossary_id": glossary_id, "error": str(e)}
|
|
)
|
|
|
|
|
|
def validate_glossary_access(glossary_id: str, user_id: str) -> bool:
|
|
"""
|
|
Validate that a glossary exists and is accessible to the user
|
|
(owner, or member of the workspace the glossary is shared with).
|
|
|
|
This is a lightweight check that doesn't return the terms,
|
|
useful for early validation before starting a translation job.
|
|
|
|
Args:
|
|
glossary_id: UUID of the glossary
|
|
user_id: UUID of the user
|
|
|
|
Returns:
|
|
True if glossary exists and is accessible
|
|
|
|
Raises:
|
|
GlossaryNotFoundError: If glossary doesn't exist or isn't accessible
|
|
"""
|
|
try:
|
|
with get_sync_session() as session:
|
|
glossary = (
|
|
session.query(Glossary).filter(Glossary.id == glossary_id).first()
|
|
)
|
|
|
|
if not glossary or not _glossary_accessible(session, glossary, user_id):
|
|
raise GlossaryNotFoundError(
|
|
message="Glossaire introuvable ou vous n'avez pas accès à cette ressource.",
|
|
details={"glossary_id": glossary_id}
|
|
)
|
|
|
|
return True
|
|
|
|
except GlossaryNotFoundError:
|
|
raise
|
|
except Exception as e:
|
|
logger.error(f"Error validating glossary access {glossary_id}: {e}")
|
|
raise GlossaryNotFoundError(
|
|
message="Erreur lors de la validation du glossaire.",
|
|
details={"glossary_id": glossary_id, "error": str(e)}
|
|
)
|
|
|
|
|
|
def format_glossary_for_prompt(
|
|
terms: List[Dict[str, str]],
|
|
source_lang: str = "fr",
|
|
target_lang: str = "en",
|
|
glossary_target_lang: str = "multi",
|
|
) -> str:
|
|
"""
|
|
Format glossary terms for injection into an LLM system prompt.
|
|
|
|
When a term has a translation for target_lang in its translations dict,
|
|
that specific translation is used. Otherwise, falls back to the default
|
|
target field (backward compat). For templates that only have EN translations,
|
|
the LLM is instructed to derive the correct target_lang equivalent.
|
|
|
|
Args:
|
|
terms: List of dicts with 'source', 'target', and optional 'translations'
|
|
source_lang: ISO code of the source language
|
|
target_lang: ISO code of the target language
|
|
glossary_target_lang: ISO code of the glossary's target language configuration
|
|
|
|
Returns:
|
|
Formatted string for LLM prompt
|
|
"""
|
|
if not terms:
|
|
return ""
|
|
|
|
sorted_terms = sorted(terms, key=lambda t: len(t.get("source", "")), reverse=True)
|
|
|
|
lines = [
|
|
f"TERMINOLOGY GLOSSARY (translate from {source_lang} to {target_lang}):",
|
|
""
|
|
]
|
|
|
|
has_fallback = False
|
|
for term in sorted_terms:
|
|
source = term.get("source", "").strip()
|
|
if not source:
|
|
continue
|
|
|
|
translations = term.get("translations", {}) or {}
|
|
specific = translations.get(target_lang, "").strip()
|
|
default_target = term.get("target", "").strip()
|
|
|
|
if specific:
|
|
source_escaped = source.replace("'", "\\'")
|
|
target_escaped = specific.replace("'", "\\'")
|
|
lines.append(f"- '{source_escaped}' → '{target_escaped}'")
|
|
elif default_target:
|
|
source_escaped = source.replace("'", "\\'")
|
|
target_escaped = default_target.replace("'", "\\'")
|
|
if glossary_target_lang == target_lang:
|
|
lines.append(f"- '{source_escaped}' → '{target_escaped}'")
|
|
else:
|
|
lines.append(f"- '{source_escaped}' → '{target_escaped}' (EN reference, adapt to {target_lang})")
|
|
has_fallback = True
|
|
# If neither specific nor default, skip the term
|
|
|
|
if not any(line.startswith("- ") for line in lines):
|
|
return ""
|
|
|
|
lines.extend([
|
|
"",
|
|
"IMPORTANT: Always use these translations when the terms appear in the text."
|
|
])
|
|
|
|
if has_fallback:
|
|
lines.append(
|
|
"NOTE: Some entries show an English reference — translate to the correct "
|
|
f"{target_lang} equivalent while preserving the intended meaning."
|
|
)
|
|
|
|
return "\n".join(lines)
|
|
|
|
|
|
def build_full_prompt(
|
|
custom_prompt: Optional[str],
|
|
glossary_terms: Optional[List[Dict[str, str]]],
|
|
source_lang: str = "fr",
|
|
target_lang: str = "en",
|
|
glossary_target_lang: str = "multi",
|
|
formality: Optional[str] = None,
|
|
) -> str:
|
|
"""
|
|
Build the complete prompt combining custom prompt, glossary, formality
|
|
and regional variant directives.
|
|
|
|
Args:
|
|
custom_prompt: Optional custom system prompt from user
|
|
glossary_terms: Optional list of glossary terms
|
|
source_lang: ISO code of the source language
|
|
target_lang: ISO code of the target language
|
|
glossary_target_lang: ISO code of the glossary's target language configuration
|
|
formality: Optional tone override — "formal" or "informal". Only
|
|
meaningful for LLM engines (ignored by classic engines).
|
|
|
|
Returns:
|
|
Combined prompt string
|
|
"""
|
|
parts = []
|
|
|
|
if custom_prompt:
|
|
parts.append(custom_prompt)
|
|
|
|
if glossary_terms:
|
|
glossary_prompt = format_glossary_for_prompt(
|
|
glossary_terms, source_lang, target_lang, glossary_target_lang
|
|
)
|
|
if glossary_prompt:
|
|
parts.append(glossary_prompt)
|
|
|
|
if formality in ("formal", "informal"):
|
|
if formality == "formal":
|
|
parts.append(
|
|
"TONE: Use a formal, professional register throughout "
|
|
"(formal address (vous/Sie) where the language distinguishes; "
|
|
"no slang, no contractions where avoidable)."
|
|
)
|
|
else:
|
|
parts.append(
|
|
"TONE: Use an informal, natural register throughout "
|
|
"(tu-style address where the language distinguishes; "
|
|
"contractions welcome)."
|
|
)
|
|
|
|
# Regional variant: when the target code carries a region (pt-BR,
|
|
# fr-CA, zh-CN...), make the expected variety explicit — LLMs default
|
|
# to the dominant variant otherwise (pt-PT, fr-FR...).
|
|
if target_lang and "-" in target_lang and target_lang != "auto":
|
|
from core.languages import language_name
|
|
|
|
name = language_name(target_lang)
|
|
if name and name != target_lang:
|
|
parts.append(
|
|
f"REGIONAL VARIANT: write specifically in {name}."
|
|
)
|
|
|
|
return "\n\n".join(parts) if parts else "" |