diff --git a/misc/management/commands/retranslate_html_links.py b/misc/management/commands/retranslate_html_links.py new file mode 100644 index 0000000000..3674f77041 --- /dev/null +++ b/misc/management/commands/retranslate_html_links.py @@ -0,0 +1,103 @@ +import logging +from typing import Any + +from django.core.management.base import BaseCommand, CommandParser +from django.db.models import Q +from modeltranslation.translator import translator +from modeltranslation.utils import build_localized_fieldname +from django.conf import settings + +from utils.translation import ( + detect_and_update_content_language, + get_translation_fields_for_model, + update_translations_for_model, +) + +logging.basicConfig(level=logging.DEBUG) + +# Markdown links whose label contains an inline HTML tag - e.g. [text](url). +# Those were destroyed by translating markdown with the (previously defaulted) +# "text/html" mimeType, which dropped the "[" and "](url)" around the tag. +BROKEN_LINK_REGEX = r"\[[^\]]*<[a-zA-Z]" + + +def get_models(app_label, model_name): + models = translator.get_registered_models(abstract=False) + models = [m for m in models if not m._meta.proxy and m._meta.managed] + + if app_label: + models = [m for m in models if m._meta.app_label == app_label] + + if model_name: + model_name = model_name.lower() + models = [m for m in models if m._meta.model_name == model_name] + + return models + + +def filter_affected(queryset): + model = queryset.model + original_lang = settings.ORIGINAL_LANGUAGE_CODE + + match_any_field = Q() + for field_name in get_translation_fields_for_model(model): + original_field = build_localized_fieldname(field_name, original_lang) + match_any_field |= Q(**{f"{original_field}__regex": BROKEN_LINK_REGEX}) + + return queryset.filter( + match_any_field, + is_automatically_translated=True, + # Objects without a detected language were never translated, so they have + # nothing broken to repair. Leave them to the regular update_translations run + # rather than translating them for the first time as a side effect of this. + content_original_lang__isnull=False, + ) + + +class Command(BaseCommand): + help = ( + "Re-translate content whose markdown links were destroyed by translating" + " with the html mimeType. Dry run by default." + ) + + def add_arguments(self, parser: CommandParser) -> None: + parser.add_argument("app_label", nargs="?") + parser.add_argument("model_name", nargs="?") + parser.add_argument("--batch_size", default=5, type=int) + parser.add_argument( + "--apply", + action="store_true", + help="Actually re-translate. Without it the command only reports counts.", + ) + + def handle(self, *args: Any, **options: Any) -> None: + batch_size = options["batch_size"] + apply = options["apply"] + + for model in get_models(options["app_label"], options["model_name"]): + pks = list( + filter_affected(model._default_manager.all()).values_list( + "pk", flat=True + ) + ) + + if not pks: + continue + + logging.info( + f"{model.__name__}: {len(pks)} affected objects, e.g. {pks[:10]}" + ) + + if not apply: + continue + + queryset = model._default_manager.filter(pk__in=pks) + # Nulling the hash is what makes queryset_filter_outdated_translations + # pick these up again - the original content itself has not changed. + queryset.update(content_last_md5=None) + + detect_and_update_content_language(queryset, batch_size) + update_translations_for_model(queryset, batch_size) + + if not apply: + logging.info("Dry run - pass --apply to re-translate.") diff --git a/tests/unit/test_utils/test_translation.py b/tests/unit/test_utils/test_translation.py new file mode 100644 index 0000000000..91ceedc919 --- /dev/null +++ b/tests/unit/test_utils/test_translation.py @@ -0,0 +1,164 @@ +import asyncio +from contextlib import asynccontextmanager +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +from utils.translation import ( + agoogle_translate_text, + rejoin_translated, + split_for_translation, +) + +# The Spanish original of question 35557, which regressed: translating it as HTML +# dropped the "[" and "](url)" around the tag, leaving unclickable text. +GEORGESCU_ES = ( + "En noviembre de 2024, Călin Georgescu, un [candidato ultranacionalista con" + " sentimientos pro-rusos](https://ro.wikipedia.org/wiki/C%C4%83lin_Georgescu)," + " ganó la primera vuelta." +) + + +def roundtrip(text, translate=lambda line: line): + translatable, layout = split_for_translation(text) + return rejoin_translated(layout, [translate(line) for line in translatable]) + + +class TestSplitForTranslation: + def test_only_non_blank_lines_are_sent(self): + translatable, _ = split_for_translation("uno\n\n\ndos") + assert translatable == ["uno", "dos"] + + def test_blank_lines_are_preserved(self): + assert roundtrip("uno\n\n\ndos") == "uno\n\n\ndos" + + def test_leading_and_trailing_newlines_are_preserved(self): + assert roundtrip("\n\nuno\n\n") == "\n\nuno\n\n" + + def test_carriage_returns_are_preserved(self): + assert roundtrip("uno\r\n\r\ndos") == "uno\r\n\r\ndos" + + def test_single_line(self): + assert roundtrip("uno") == "uno" + + def test_empty_text_has_nothing_to_translate(self): + translatable, layout = split_for_translation("") + assert translatable == [] + assert rejoin_translated(layout, []) == "" + + def test_whitespace_only_text_is_untouched(self): + translatable, _ = split_for_translation("\n \n") + assert translatable == [] + assert roundtrip("\n \n") == "\n \n" + + def test_markdown_structure_survives_a_translation(self): + text = "# Título\n\nUn [enlace](https://example.com) aquí.\n\n* uno\n* dos" + assert roundtrip(text, translate=str.upper) == ( + "# TÍTULO\n\nUN [ENLACE](HTTPS://EXAMPLE.COM) AQUÍ.\n\n* UNO\n* DOS" + ) + + def test_link_with_inline_html_label_is_left_intact(self): + # The regression from #2299: the link must reach the API in one piece. + translatable, _ = split_for_translation(GEORGESCU_ES) + assert translatable == [GEORGESCU_ES] + assert roundtrip(GEORGESCU_ES) == GEORGESCU_ES + + +class TestRejoinTranslated: + def test_rejects_a_short_response(self): + _, layout = split_for_translation("uno\ndos") + with pytest.raises(ValueError): + rejoin_translated(layout, ["one"]) + + +def mock_translate_session(translations): + response = MagicMock() + response.status = 200 + response.json = AsyncMock( + return_value={"translations": [{"translatedText": t} for t in translations]} + ) + + post_calls = [] + + @asynccontextmanager + async def post(url, headers, json): + post_calls.append({"url": url, "headers": headers, "json": json}) + yield response + + session = MagicMock() + session.post = post + + @asynccontextmanager + async def client_session(): + yield session + + return client_session, post_calls + + +class TestAgoogleTranslateText: + def _translate(self, text, translations): + client_session, post_calls = mock_translate_session(translations) + + with ( + patch( + "utils.translation.get_and_cache_sa_info", + return_value=("token", "project"), + ), + patch("aiohttp.ClientSession", client_session), + ): + result = asyncio.run(agoogle_translate_text("es", "en", text)) + + return result, post_calls + + def test_sends_plain_text_mimetype_in_the_payload(self): + _, post_calls = self._translate("hola", ["hello"]) + + payload = post_calls[0]["json"] + assert payload["mimeType"] == "text/plain" + # mimeType in the headers is what caused #2299 - it was silently ignored. + assert "mimeType" not in post_calls[0]["headers"] + + def test_sends_contents_as_a_list_of_lines(self): + _, post_calls = self._translate("hola\n\nadiós", ["hello", "goodbye"]) + + assert post_calls[0]["json"]["contents"] == ["hola", "adiós"] + + def test_reassembles_the_translated_lines(self): + result, _ = self._translate("hola\n\nadiós", ["hello", "goodbye"]) + + assert result == "hello\n\ngoodbye" + + def test_unescapes_html_entities_in_the_response(self): + result, _ = self._translate("Tom y Jerry", ["Tom & Jerry"]) + + assert result == "Tom & Jerry" + + def test_same_language_skips_the_api(self): + client_session, post_calls = mock_translate_session([]) + + with ( + patch( + "utils.translation.get_and_cache_sa_info", + return_value=("token", "project"), + ), + patch("aiohttp.ClientSession", client_session), + ): + result = asyncio.run(agoogle_translate_text("en", "en", "unchanged")) + + assert result == "unchanged" + assert post_calls == [] + + def test_text_without_anything_translatable_skips_the_api(self): + client_session, post_calls = mock_translate_session([]) + + with ( + patch( + "utils.translation.get_and_cache_sa_info", + return_value=("token", "project"), + ), + patch("aiohttp.ClientSession", client_session), + ): + result = asyncio.run(agoogle_translate_text("es", "en", "\n\n")) + + assert result == "\n\n" + assert post_calls == [] diff --git a/utils/translation.py b/utils/translation.py index 354527ebae..0895e47d63 100644 --- a/utils/translation.py +++ b/utils/translation.py @@ -76,41 +76,77 @@ async def agoogle_translate_detect_language(text): raise Exception(f"Error detecting language: {error}") +def split_for_translation(text): + """ + Splits text into the lines that need translating, plus a layout describing how + to put it back together. Each layout entry is either None (a placeholder for the + next translated line) or a literal string to emit as-is (blank lines). + + Translating line by line keeps the line structure intact no matter how the + translation service handles whitespace, and lines map onto markdown paragraphs. + """ + layout = [] + translatable = [] + + for line in text.split("\n"): + if line.strip(): + layout.append(None) + translatable.append(line) + else: + layout.append(line) + + return translatable, layout + + +def rejoin_translated(layout, translated): + expected = sum(1 for entry in layout if entry is None) + if len(translated) != expected: + raise ValueError(f"Expected {expected} translated lines, got {len(translated)}") + + translated_iter = iter(translated) + return "\n".join( + next(translated_iter) if entry is None else entry for entry in layout + ) + + async def agoogle_translate_text(source_language, target_language, text): token, project_id = get_and_cache_sa_info() if source_language == target_language: return text - # Google Translates doesn't preserve new lines, as it translates text as if it was HTML. - # so we use a hack by inserting a
element for each \n before translating, - # and then we replace it back with the new line after translation. - # In addition, it also html-escapes the input so we need to unescape it after the translation - text = text.replace("\n", "
") + translatable, layout = split_for_translation(text) + if not translatable: + return text + url = f"https://translation.googleapis.com/v3/projects/{project_id}:translateText" headers = { "Authorization": f"Bearer {token}", "x-goog-user-project": project_id, "Content-Type": "application/json; charset=utf-8", - "mimeType": "text/plain", } + # mimeType must be part of the payload, not of the headers. It defaults to + # "text/html", which makes the service parse our markdown as HTML: inline tags + # split it into separate text nodes and the non-prose ones (like the "](url)" + # of a markdown link) get dropped, silently destroying links. data = { "sourceLanguageCode": source_language, "targetLanguageCode": target_language, - "contents": text, + "mimeType": "text/plain", + "contents": translatable, } async with aiohttp.ClientSession() as session: async with session.post(url, headers=headers, json=data) as response: if response.status == 200: result = await response.json() - output = result["translations"][0]["translatedText"] - # See the comment above with the new lines - output = output.replace("
", "\n") - output = html.unescape(output) - return output + translated = [t["translatedText"] for t in result["translations"]] + # Kept from the text/html days: harmless if the response is already + # unescaped, and it still corrects the output if it is not. + translated = [html.unescape(t) for t in translated] + return rejoin_translated(layout, translated) else: error = await response.text() raise Exception(f"Error translating text: {error}")