Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
103 changes: 103 additions & 0 deletions misc/management/commands/retranslate_html_links.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,103 @@
import logging
from typing import Any

from django.core.management.base import BaseCommand, CommandParser
from django.db.models import Q
from modeltranslation.translator import translator
from modeltranslation.utils import build_localized_fieldname
from django.conf import settings

from utils.translation import (
detect_and_update_content_language,
get_translation_fields_for_model,
update_translations_for_model,
)

logging.basicConfig(level=logging.DEBUG)

# Markdown links whose label contains an inline HTML tag - e.g. [<u>text</u>](url).
# Those were destroyed by translating markdown with the (previously defaulted)
# "text/html" mimeType, which dropped the "[" and "](url)" around the tag.
BROKEN_LINK_REGEX = r"\[[^\]]*<[a-zA-Z]"


def get_models(app_label, model_name):
models = translator.get_registered_models(abstract=False)
models = [m for m in models if not m._meta.proxy and m._meta.managed]

if app_label:
models = [m for m in models if m._meta.app_label == app_label]

if model_name:
model_name = model_name.lower()
models = [m for m in models if m._meta.model_name == model_name]

return models


def filter_affected(queryset):
model = queryset.model
original_lang = settings.ORIGINAL_LANGUAGE_CODE

match_any_field = Q()
for field_name in get_translation_fields_for_model(model):
original_field = build_localized_fieldname(field_name, original_lang)
match_any_field |= Q(**{f"{original_field}__regex": BROKEN_LINK_REGEX})

return queryset.filter(
match_any_field,
is_automatically_translated=True,
# Objects without a detected language were never translated, so they have
# nothing broken to repair. Leave them to the regular update_translations run
# rather than translating them for the first time as a side effect of this.
content_original_lang__isnull=False,
)


class Command(BaseCommand):
help = (
"Re-translate content whose markdown links were destroyed by translating"
" with the html mimeType. Dry run by default."
)

def add_arguments(self, parser: CommandParser) -> None:
parser.add_argument("app_label", nargs="?")
parser.add_argument("model_name", nargs="?")
parser.add_argument("--batch_size", default=5, type=int)
parser.add_argument(
"--apply",
action="store_true",
help="Actually re-translate. Without it the command only reports counts.",
)

def handle(self, *args: Any, **options: Any) -> None:
batch_size = options["batch_size"]
apply = options["apply"]

for model in get_models(options["app_label"], options["model_name"]):
pks = list(
filter_affected(model._default_manager.all()).values_list(
"pk", flat=True
)
)

if not pks:
continue

logging.info(
f"{model.__name__}: {len(pks)} affected objects, e.g. {pks[:10]}"
)

if not apply:
continue

queryset = model._default_manager.filter(pk__in=pks)
# Nulling the hash is what makes queryset_filter_outdated_translations
# pick these up again - the original content itself has not changed.
queryset.update(content_last_md5=None)

detect_and_update_content_language(queryset, batch_size)
update_translations_for_model(queryset, batch_size)

if not apply:
logging.info("Dry run - pass --apply to re-translate.")
164 changes: 164 additions & 0 deletions tests/unit/test_utils/test_translation.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,164 @@
import asyncio
from contextlib import asynccontextmanager
from unittest.mock import AsyncMock, MagicMock, patch

import pytest

from utils.translation import (
agoogle_translate_text,
rejoin_translated,
split_for_translation,
)

# The Spanish original of question 35557, which regressed: translating it as HTML
# dropped the "[" and "](url)" around the <u> tag, leaving unclickable text.
GEORGESCU_ES = (
"En noviembre de 2024, Călin Georgescu, un [<u>candidato ultranacionalista con"
" sentimientos pro-rusos</u>](https://ro.wikipedia.org/wiki/C%C4%83lin_Georgescu),"
" ganó la primera vuelta."
)


def roundtrip(text, translate=lambda line: line):
translatable, layout = split_for_translation(text)
return rejoin_translated(layout, [translate(line) for line in translatable])


class TestSplitForTranslation:
def test_only_non_blank_lines_are_sent(self):
translatable, _ = split_for_translation("uno\n\n\ndos")
assert translatable == ["uno", "dos"]

def test_blank_lines_are_preserved(self):
assert roundtrip("uno\n\n\ndos") == "uno\n\n\ndos"

def test_leading_and_trailing_newlines_are_preserved(self):
assert roundtrip("\n\nuno\n\n") == "\n\nuno\n\n"

def test_carriage_returns_are_preserved(self):
assert roundtrip("uno\r\n\r\ndos") == "uno\r\n\r\ndos"

def test_single_line(self):
assert roundtrip("uno") == "uno"

def test_empty_text_has_nothing_to_translate(self):
translatable, layout = split_for_translation("")
assert translatable == []
assert rejoin_translated(layout, []) == ""

def test_whitespace_only_text_is_untouched(self):
translatable, _ = split_for_translation("\n \n")
assert translatable == []
assert roundtrip("\n \n") == "\n \n"

def test_markdown_structure_survives_a_translation(self):
text = "# Título\n\nUn [enlace](https://example.com) aquí.\n\n* uno\n* dos"
assert roundtrip(text, translate=str.upper) == (
"# TÍTULO\n\nUN [ENLACE](HTTPS://EXAMPLE.COM) AQUÍ.\n\n* UNO\n* DOS"
)

def test_link_with_inline_html_label_is_left_intact(self):
# The regression from #2299: the link must reach the API in one piece.
translatable, _ = split_for_translation(GEORGESCU_ES)
assert translatable == [GEORGESCU_ES]
assert roundtrip(GEORGESCU_ES) == GEORGESCU_ES


class TestRejoinTranslated:
def test_rejects_a_short_response(self):
_, layout = split_for_translation("uno\ndos")
with pytest.raises(ValueError):
rejoin_translated(layout, ["one"])


def mock_translate_session(translations):
response = MagicMock()
response.status = 200
response.json = AsyncMock(
return_value={"translations": [{"translatedText": t} for t in translations]}
)

post_calls = []

@asynccontextmanager
async def post(url, headers, json):
post_calls.append({"url": url, "headers": headers, "json": json})
yield response

session = MagicMock()
session.post = post

@asynccontextmanager
async def client_session():
yield session

return client_session, post_calls


class TestAgoogleTranslateText:
def _translate(self, text, translations):
client_session, post_calls = mock_translate_session(translations)

with (
patch(
"utils.translation.get_and_cache_sa_info",
return_value=("token", "project"),
),
patch("aiohttp.ClientSession", client_session),
):
result = asyncio.run(agoogle_translate_text("es", "en", text))

return result, post_calls

def test_sends_plain_text_mimetype_in_the_payload(self):
_, post_calls = self._translate("hola", ["hello"])

payload = post_calls[0]["json"]
assert payload["mimeType"] == "text/plain"
# mimeType in the headers is what caused #2299 - it was silently ignored.
assert "mimeType" not in post_calls[0]["headers"]

def test_sends_contents_as_a_list_of_lines(self):
_, post_calls = self._translate("hola\n\nadiós", ["hello", "goodbye"])

assert post_calls[0]["json"]["contents"] == ["hola", "adiós"]

def test_reassembles_the_translated_lines(self):
result, _ = self._translate("hola\n\nadiós", ["hello", "goodbye"])

assert result == "hello\n\ngoodbye"

def test_unescapes_html_entities_in_the_response(self):
result, _ = self._translate("Tom y Jerry", ["Tom &amp; Jerry"])

assert result == "Tom & Jerry"

def test_same_language_skips_the_api(self):
client_session, post_calls = mock_translate_session([])

with (
patch(
"utils.translation.get_and_cache_sa_info",
return_value=("token", "project"),
),
patch("aiohttp.ClientSession", client_session),
):
result = asyncio.run(agoogle_translate_text("en", "en", "unchanged"))

assert result == "unchanged"
assert post_calls == []

def test_text_without_anything_translatable_skips_the_api(self):
client_session, post_calls = mock_translate_session([])

with (
patch(
"utils.translation.get_and_cache_sa_info",
return_value=("token", "project"),
),
patch("aiohttp.ClientSession", client_session),
):
result = asyncio.run(agoogle_translate_text("es", "en", "\n\n"))

assert result == "\n\n"
assert post_calls == []
60 changes: 48 additions & 12 deletions utils/translation.py
Original file line number Diff line number Diff line change
Expand Up @@ -76,41 +76,77 @@ async def agoogle_translate_detect_language(text):
raise Exception(f"Error detecting language: {error}")


def split_for_translation(text):
"""
Splits text into the lines that need translating, plus a layout describing how
to put it back together. Each layout entry is either None (a placeholder for the
next translated line) or a literal string to emit as-is (blank lines).

Translating line by line keeps the line structure intact no matter how the
translation service handles whitespace, and lines map onto markdown paragraphs.
"""
layout = []
translatable = []

for line in text.split("\n"):
if line.strip():
layout.append(None)
translatable.append(line)
else:
layout.append(line)

return translatable, layout


def rejoin_translated(layout, translated):
expected = sum(1 for entry in layout if entry is None)
if len(translated) != expected:
raise ValueError(f"Expected {expected} translated lines, got {len(translated)}")

translated_iter = iter(translated)
return "\n".join(
next(translated_iter) if entry is None else entry for entry in layout
)


async def agoogle_translate_text(source_language, target_language, text):
token, project_id = get_and_cache_sa_info()

if source_language == target_language:
return text

# Google Translates doesn't preserve new lines, as it translates text as if it was HTML.
# so we use a hack by inserting a <br> element for each \n before translating,
# and then we replace it back with the new line after translation.
# In addition, it also html-escapes the input so we need to unescape it after the translation
text = text.replace("\n", "<br>")
translatable, layout = split_for_translation(text)
if not translatable:
return text

url = f"https://translation.googleapis.com/v3/projects/{project_id}:translateText"

headers = {
"Authorization": f"Bearer {token}",
"x-goog-user-project": project_id,
"Content-Type": "application/json; charset=utf-8",
"mimeType": "text/plain",
}

# mimeType must be part of the payload, not of the headers. It defaults to
# "text/html", which makes the service parse our markdown as HTML: inline tags
# split it into separate text nodes and the non-prose ones (like the "](url)"
# of a markdown link) get dropped, silently destroying links.
data = {
"sourceLanguageCode": source_language,
"targetLanguageCode": target_language,
"contents": text,
"mimeType": "text/plain",
"contents": translatable,
}

async with aiohttp.ClientSession() as session:
async with session.post(url, headers=headers, json=data) as response:
if response.status == 200:
result = await response.json()
output = result["translations"][0]["translatedText"]
# See the comment above with the new lines
output = output.replace("<br>", "\n")
output = html.unescape(output)
return output
translated = [t["translatedText"] for t in result["translations"]]
# Kept from the text/html days: harmless if the response is already
# unescaped, and it still corrects the output if it is not.
translated = [html.unescape(t) for t in translated]
return rejoin_translated(layout, translated)
else:
error = await response.text()
raise Exception(f"Error translating text: {error}")
Expand Down
Loading