Files
office_translator/services/glossary_service.py
sepehr b4e873ad2c
Some checks failed
Deploy to Production / Build and Deploy (push) Failing after 2m14s
feat(review,teams): review foundation — segments, side-by-side editor, rebuild, XLIFF, team workspaces
Foundations:
- TranslationSegment model + migration f7e8d9c0b1a2 (segments, workspaces,
  workspace_members, glossaries.workspace_id)
- SegmentRecorder injected into all 4 translators: unique (source,
  translation) pairs captured per job and persisted (best-effort)
- set_segment_overrides: human-reviewed translations applied verbatim on
  rebuild — top priority over TM and provider, zero API calls

Review API (routes/review_routes.py):
- GET /translations/{id}/segments (owner or job token)
- PATCH /segments/{id} edit/approve — feeds the per-user TM so approved
  translations are reused in later jobs
- POST /translations/{id}/rebuild — rebuild document with reviewed text
- GET/POST /translations/{id}/xliff — XLIFF 1.2 export/import (edited
  segments export their reviewed text)

Review editor (frontend /dashboard/reviews/[jobId]):
- side-by-side source/translation table, inline edit, approve (single or
  all), rebuild & download (auth blob), XLIFF export/import, 13 locales
- 'Relire et corriger' link on the translation-complete screen

Team workspaces (routes/workspace_routes.py + /dashboard/teams):
- Workspace/WorkspaceMember models, roles owner/admin/member
- create (Business plan), list with seat usage, invite by email with
  seat-limit enforcement (Business=5, Enterprise unlimited), removal
- shared glossaries: workspace members can use a glossary shared to their
  workspace (access check extended)

Tests: 1184 passed / 0 failed (11 new: recorder, overrides, docx
capture->rebuild e2e, XLIFF structure/escaping, seats, workspace CRUD,
shared glossary access)
2026-08-29 19:04:32 +02:00

276 lines
9.4 KiB
Python

"""
Glossary Service for Translation
Story 3.10: Glossaires - Application lors Traduction LLM
Provides functions to retrieve glossary terms and format them for LLM prompts.
"""
import logging
from typing import List, Dict, Any, Optional
from database.connection import get_sync_session
from database.models import Glossary, GlossaryTerm
from utils.exceptions import GlossaryNotFoundError
logger = logging.getLogger(__name__)
def _user_workspace_ids(session, user_id: str) -> List[str]:
"""Workspaces the user belongs to (for shared-glossary access)."""
from database.models import WorkspaceMember
rows = (
session.query(WorkspaceMember.workspace_id)
.filter(WorkspaceMember.user_id == user_id)
.all()
)
return [r[0] for r in rows]
def _glossary_accessible(session, glossary: Glossary, user_id: str) -> bool:
"""Owner always; otherwise any workspace the glossary is shared with."""
if str(glossary.user_id) == str(user_id):
return True
if glossary.workspace_id:
return glossary.workspace_id in _user_workspace_ids(session, user_id)
return False
def get_glossary_terms(glossary_id: str, user_id: str) -> Dict[str, Any]:
"""
Retrieve glossary terms and metadata for a glossary the user can access
(owner, or member of the workspace the glossary is shared with).
Args:
glossary_id: UUID of the glossary
user_id: UUID of the user
Returns:
Dict with 'source_language' and 'terms' (list of dicts with source, target, translations)
Raises:
GlossaryNotFoundError: If glossary doesn't exist or isn't accessible
"""
try:
with get_sync_session() as session:
glossary = (
session.query(Glossary).filter(Glossary.id == glossary_id).first()
)
if not glossary or not _glossary_accessible(session, glossary, user_id):
raise GlossaryNotFoundError(
message="Glossaire introuvable ou vous n'avez pas accès à cette ressource.",
details={"glossary_id": glossary_id}
)
terms = (
session.query(GlossaryTerm)
.filter(GlossaryTerm.glossary_id == glossary_id)
.all()
)
result = [{
"source": term.source,
"target": term.target,
"translations": term.translations or {}
} for term in terms]
logger.info(
f"Retrieved {len(result)} terms from glossary {glossary_id} for user {user_id}"
)
return {
"source_language": glossary.source_language or "fr",
"target_language": getattr(glossary, "target_language", None) or "multi",
"terms": result,
}
except GlossaryNotFoundError:
raise
except Exception as e:
logger.error(f"Error retrieving glossary {glossary_id}: {e}")
raise GlossaryNotFoundError(
message="Erreur lors de la récupération du glossaire.",
details={"glossary_id": glossary_id, "error": str(e)}
)
def validate_glossary_access(glossary_id: str, user_id: str) -> bool:
"""
Validate that a glossary exists and is accessible to the user
(owner, or member of the workspace the glossary is shared with).
This is a lightweight check that doesn't return the terms,
useful for early validation before starting a translation job.
Args:
glossary_id: UUID of the glossary
user_id: UUID of the user
Returns:
True if glossary exists and is accessible
Raises:
GlossaryNotFoundError: If glossary doesn't exist or isn't accessible
"""
try:
with get_sync_session() as session:
glossary = (
session.query(Glossary).filter(Glossary.id == glossary_id).first()
)
if not glossary or not _glossary_accessible(session, glossary, user_id):
raise GlossaryNotFoundError(
message="Glossaire introuvable ou vous n'avez pas accès à cette ressource.",
details={"glossary_id": glossary_id}
)
return True
except GlossaryNotFoundError:
raise
except Exception as e:
logger.error(f"Error validating glossary access {glossary_id}: {e}")
raise GlossaryNotFoundError(
message="Erreur lors de la validation du glossaire.",
details={"glossary_id": glossary_id, "error": str(e)}
)
def format_glossary_for_prompt(
terms: List[Dict[str, str]],
source_lang: str = "fr",
target_lang: str = "en",
glossary_target_lang: str = "multi",
) -> str:
"""
Format glossary terms for injection into an LLM system prompt.
When a term has a translation for target_lang in its translations dict,
that specific translation is used. Otherwise, falls back to the default
target field (backward compat). For templates that only have EN translations,
the LLM is instructed to derive the correct target_lang equivalent.
Args:
terms: List of dicts with 'source', 'target', and optional 'translations'
source_lang: ISO code of the source language
target_lang: ISO code of the target language
glossary_target_lang: ISO code of the glossary's target language configuration
Returns:
Formatted string for LLM prompt
"""
if not terms:
return ""
sorted_terms = sorted(terms, key=lambda t: len(t.get("source", "")), reverse=True)
lines = [
f"TERMINOLOGY GLOSSARY (translate from {source_lang} to {target_lang}):",
""
]
has_fallback = False
for term in sorted_terms:
source = term.get("source", "").strip()
if not source:
continue
translations = term.get("translations", {}) or {}
specific = translations.get(target_lang, "").strip()
default_target = term.get("target", "").strip()
if specific:
source_escaped = source.replace("'", "\\'")
target_escaped = specific.replace("'", "\\'")
lines.append(f"- '{source_escaped}''{target_escaped}'")
elif default_target:
source_escaped = source.replace("'", "\\'")
target_escaped = default_target.replace("'", "\\'")
if glossary_target_lang == target_lang:
lines.append(f"- '{source_escaped}''{target_escaped}'")
else:
lines.append(f"- '{source_escaped}''{target_escaped}' (EN reference, adapt to {target_lang})")
has_fallback = True
# If neither specific nor default, skip the term
if not any(line.startswith("- ") for line in lines):
return ""
lines.extend([
"",
"IMPORTANT: Always use these translations when the terms appear in the text."
])
if has_fallback:
lines.append(
"NOTE: Some entries show an English reference — translate to the correct "
f"{target_lang} equivalent while preserving the intended meaning."
)
return "\n".join(lines)
def build_full_prompt(
custom_prompt: Optional[str],
glossary_terms: Optional[List[Dict[str, str]]],
source_lang: str = "fr",
target_lang: str = "en",
glossary_target_lang: str = "multi",
formality: Optional[str] = None,
) -> str:
"""
Build the complete prompt combining custom prompt, glossary, formality
and regional variant directives.
Args:
custom_prompt: Optional custom system prompt from user
glossary_terms: Optional list of glossary terms
source_lang: ISO code of the source language
target_lang: ISO code of the target language
glossary_target_lang: ISO code of the glossary's target language configuration
formality: Optional tone override — "formal" or "informal". Only
meaningful for LLM engines (ignored by classic engines).
Returns:
Combined prompt string
"""
parts = []
if custom_prompt:
parts.append(custom_prompt)
if glossary_terms:
glossary_prompt = format_glossary_for_prompt(
glossary_terms, source_lang, target_lang, glossary_target_lang
)
if glossary_prompt:
parts.append(glossary_prompt)
if formality in ("formal", "informal"):
if formality == "formal":
parts.append(
"TONE: Use a formal, professional register throughout "
"(formal address (vous/Sie) where the language distinguishes; "
"no slang, no contractions where avoidable)."
)
else:
parts.append(
"TONE: Use an informal, natural register throughout "
"(tu-style address where the language distinguishes; "
"contractions welcome)."
)
# Regional variant: when the target code carries a region (pt-BR,
# fr-CA, zh-CN...), make the expected variety explicit — LLMs default
# to the dominant variant otherwise (pt-PT, fr-FR...).
if target_lang and "-" in target_lang and target_lang != "auto":
from core.languages import language_name
name = language_name(target_lang)
if name and name != target_lang:
parts.append(
f"REGIONAL VARIANT: write specifically in {name}."
)
return "\n\n".join(parts) if parts else ""