Merge pull request #2869 from davotoula/fix/1180-translation-image-url
Preserve image URLs in CJK translations (#1180)
This commit is contained in:
@@ -167,6 +167,20 @@ class TranslationsTest {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun testJapaneseImageUrlIssue1180() {
|
||||||
|
// Regression for https://github.com/vitorpamplona/amethyst/issues/1180 — image URL at the end
|
||||||
|
// of a Japanese post must survive ja→en auto-translation intact.
|
||||||
|
assertTranslateContains(
|
||||||
|
"https://image.nostr.build/3cd07f308c13dd9197d03430a75bf5326f86d2ab68674af971e915f55a638ecd.jpg",
|
||||||
|
"「正方形のカードを並べて盤面にして、その上でカードゲームとボードゲーム同時にやったらおもろくね?」" +
|
||||||
|
"という誰でも思いつきはしそうなアイデアを、ちゃんとルール練り上げて実装まで漕ぎ着けてしまったやつ。" +
|
||||||
|
"ブースの人が「こんなところに気を遣ったんですわ〜」っていうのに全部ウンウンわかるそれそれと思わされたので買った " +
|
||||||
|
"https://image.nostr.build/3cd07f308c13dd9197d03430a75bf5326f86d2ab68674af971e915f55a638ecd.jpg ",
|
||||||
|
"en",
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
fun testHttp() {
|
fun testHttp() {
|
||||||
val text = "https://m.primal.net/MdDd.png \nRunning... \uD83D\uDE01 nostr:npub126ntw5mnermmj0znhjhgdk8lh2af72sm8qfzq48umdlnhaj9kuns3le9ll nostr:npub1getal6ykt05fsz5nqu4uld09nfj3y3qxmv8crys4aeut53unfvlqr80nfm"
|
val text = "https://m.primal.net/MdDd.png \nRunning... \uD83D\uDE01 nostr:npub126ntw5mnermmj0znhjhgdk8lh2af72sm8qfzq48umdlnhaj9kuns3le9ll nostr:npub1getal6ykt05fsz5nqu4uld09nfj3y3qxmv8crys4aeut53unfvlqr80nfm"
|
||||||
|
|||||||
+30
-10
@@ -25,16 +25,17 @@ import java.util.regex.Pattern
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Pure-JVM helpers that protect non-translatable substrings (URLs, Lightning invoices, NIP-19
|
* Pure-JVM helpers that protect non-translatable substrings (URLs, Lightning invoices, NIP-19
|
||||||
* references, NIP-08 positional references) by swapping them with single Unicode Private Use Area
|
* references, NIP-08 positional references) by swapping them with ICU MessageFormat positional
|
||||||
* codepoints around a translator. Extracted out of [LanguageTranslatorService] so the round-trip
|
* tokens ({0}, {1}, ...) around a translator. Extracted out of [LanguageTranslatorService] so the
|
||||||
* can be unit-tested without ML Kit / Android runtime.
|
* round-trip can be unit-tested without ML Kit / Android runtime.
|
||||||
*/
|
*/
|
||||||
internal object TranslationDictionary {
|
internal object TranslationDictionary {
|
||||||
// Range U+E000..U+F8FF gives 6400 placeholder slots. PUA codepoints don't appear in normal
|
// Placeholders use ICU MessageFormat positional syntax: {0}, {1}, .... ML Kit's translation
|
||||||
// user text, the translator has no rule for them so it passes them through, and using one
|
// models are trained on localization corpora and preserve these tokens intact across all source
|
||||||
// codepoint per placeholder means the translator can't split or reorder it.
|
// languages we measured (ja, zh, ko, ar, hi, ru, pt, de). Prior schemes failed: U+E000+ PUA
|
||||||
const val PLACEHOLDER_BASE: Int = 0xE000
|
// codepoints were dropped by CJK models (issue #1180); "B0/C0/A0" markers collided with user
|
||||||
const val PLACEHOLDER_LIMIT: Int = 0xF8FF - PLACEHOLDER_BASE
|
// text. 10k slots per note is effectively unbounded.
|
||||||
|
const val PLACEHOLDER_LIMIT: Int = 9999
|
||||||
|
|
||||||
// Texts shorter than this, or with no letter codepoints (emoji-only, punctuation), are skipped
|
// Texts shorter than this, or with no letter codepoints (emoji-only, punctuation), are skipped
|
||||||
// before any ML Kit work — language identification is unreliable on them.
|
// before any ML Kit work — language identification is unreliable on them.
|
||||||
@@ -51,6 +52,11 @@ internal object TranslationDictionary {
|
|||||||
// brackets ("# [0]"), so we shield them via the placeholder dictionary.
|
// brackets ("# [0]"), so we shield them via the placeholder dictionary.
|
||||||
val nip08RefRegex: Pattern = Pattern.compile("#\\[\\d+]")
|
val nip08RefRegex: Pattern = Pattern.compile("#\\[\\d+]")
|
||||||
|
|
||||||
|
// Matches placeholders the user already wrote in their note (e.g. a code snippet containing
|
||||||
|
// `{0}`). We start our own counter above the max user-supplied index so encode/decode can't
|
||||||
|
// clobber it.
|
||||||
|
private val placeholderRegex: Pattern = Pattern.compile("\\{(\\d+)\\}")
|
||||||
|
|
||||||
fun isWorthTranslating(text: String): Boolean {
|
fun isWorthTranslating(text: String): Boolean {
|
||||||
if (text.length < MIN_TRANSLATABLE_LENGTH) return false
|
if (text.length < MIN_TRANSLATABLE_LENGTH) return false
|
||||||
for (cp in text.codePoints()) {
|
for (cp in text.codePoints()) {
|
||||||
@@ -61,7 +67,7 @@ internal object TranslationDictionary {
|
|||||||
|
|
||||||
fun build(text: String): Map<String, String> {
|
fun build(text: String): Map<String, String> {
|
||||||
val dict = LinkedHashMap<String, String>()
|
val dict = LinkedHashMap<String, String>()
|
||||||
var counter = 0
|
var counter = firstFreeIndex(text)
|
||||||
|
|
||||||
fun addUnique(value: String) {
|
fun addUnique(value: String) {
|
||||||
if (value.isEmpty()) return
|
if (value.isEmpty()) return
|
||||||
@@ -84,6 +90,20 @@ internal object TranslationDictionary {
|
|||||||
return dict
|
return dict
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private fun firstFreeIndex(text: String): Int {
|
||||||
|
if (text.indexOf('{') < 0) return 0
|
||||||
|
var max = -1
|
||||||
|
val matcher = placeholderRegex.matcher(text)
|
||||||
|
while (matcher.find()) {
|
||||||
|
val idx = matcher.group(1)?.toIntOrNull() ?: continue
|
||||||
|
// Clamp at the table limit so adversarial user text like `{99999}` can't push our
|
||||||
|
// counter past PLACEHOLDER_LIMIT and make every subsequent placeholder() throw.
|
||||||
|
if (idx > PLACEHOLDER_LIMIT) continue
|
||||||
|
if (idx > max) max = idx
|
||||||
|
}
|
||||||
|
return max + 1
|
||||||
|
}
|
||||||
|
|
||||||
private inline fun Pattern.forEachMatch(
|
private inline fun Pattern.forEachMatch(
|
||||||
text: String,
|
text: String,
|
||||||
block: (String) -> Unit,
|
block: (String) -> Unit,
|
||||||
@@ -119,6 +139,6 @@ internal object TranslationDictionary {
|
|||||||
|
|
||||||
fun placeholder(index: Int): String {
|
fun placeholder(index: Int): String {
|
||||||
require(index in 0..PLACEHOLDER_LIMIT) { "placeholder index $index out of range" }
|
require(index in 0..PLACEHOLDER_LIMIT) { "placeholder index $index out of range" }
|
||||||
return String(Character.toChars(PLACEHOLDER_BASE + index))
|
return "{$index}"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+56
-13
@@ -60,14 +60,11 @@ class TranslationDictionaryTest {
|
|||||||
// ----- placeholder -----
|
// ----- placeholder -----
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
fun `placeholder is a single Unicode Private Use Area codepoint`() {
|
fun `placeholder is the ICU MessageFormat positional token for that index`() {
|
||||||
val p0 = TranslationDictionary.placeholder(0)
|
assertEquals("{0}", TranslationDictionary.placeholder(0))
|
||||||
val p1 = TranslationDictionary.placeholder(1)
|
assertEquals("{1}", TranslationDictionary.placeholder(1))
|
||||||
assertEquals(1, p0.codePointCount(0, p0.length))
|
assertEquals("{42}", TranslationDictionary.placeholder(42))
|
||||||
assertEquals(1, p1.codePointCount(0, p1.length))
|
assertNotEquals(TranslationDictionary.placeholder(0), TranslationDictionary.placeholder(1))
|
||||||
assertEquals(0xE000, p0.codePointAt(0))
|
|
||||||
assertEquals(0xE001, p1.codePointAt(0))
|
|
||||||
assertNotEquals(p0, p1)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -167,7 +164,7 @@ class TranslationDictionaryTest {
|
|||||||
val encoded = TranslationDictionary.encode(text, dict)
|
val encoded = TranslationDictionary.encode(text, dict)
|
||||||
|
|
||||||
assertFalse("URL must be removed from encoded text", encoded.contains("https://t.me/mygroup"))
|
assertFalse("URL must be removed from encoded text", encoded.contains("https://t.me/mygroup"))
|
||||||
assertTrue("encoded text must contain the placeholder", encoded.codePoints().anyMatch { it in 0xE000..0xF8FF })
|
assertTrue("encoded text must contain a {N} placeholder", Regex("\\{\\d+}").containsMatchIn(encoded))
|
||||||
|
|
||||||
val decoded = TranslationDictionary.decode(encoded, dict)
|
val decoded = TranslationDictionary.decode(encoded, dict)
|
||||||
assertEquals(text, decoded)
|
assertEquals(text, decoded)
|
||||||
@@ -185,7 +182,11 @@ class TranslationDictionaryTest {
|
|||||||
val decoded = TranslationDictionary.decode(translated, dict)!!
|
val decoded = TranslationDictionary.decode(translated, dict)!!
|
||||||
assertTrue("decoded must contain #[0]", decoded.contains("#[0]"))
|
assertTrue("decoded must contain #[0]", decoded.contains("#[0]"))
|
||||||
assertTrue("decoded must contain the nostr ref", decoded.contains("nostr:nevent1qqsabcdefgh023456"))
|
assertTrue("decoded must contain the nostr ref", decoded.contains("nostr:nevent1qqsabcdefgh023456"))
|
||||||
assertFalse("decoded must not leak placeholder codepoints", decoded.codePoints().anyMatch { it in 0xE000..0xF8FF })
|
// The dictionary picks indices above any user-supplied {N}; user text here has none, so the
|
||||||
|
// dict starts at {0}. After decode, no {N} placeholder we issued should remain.
|
||||||
|
for (token in dict.keys) {
|
||||||
|
assertFalse("decoded leaked placeholder $token", decoded.contains(token))
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -213,9 +214,9 @@ class TranslationDictionaryTest {
|
|||||||
|
|
||||||
@Test
|
@Test
|
||||||
fun `decode does not corrupt user text containing the old B0 C0 A0 placeholders`() {
|
fun `decode does not corrupt user text containing the old B0 C0 A0 placeholders`() {
|
||||||
// Regression for the pre-rewrite bug: old placeholders "B0", "C0", "A0" collided with arbitrary
|
// Regression for a pre-rewrite bug where placeholders were "B0", "C0", "A0" — those tokens
|
||||||
// user content. The new PUA placeholders are invisible codepoints that cannot occur in normal text,
|
// collided with arbitrary user content. Today's {N} tokens can't collide with B0/C0/A0
|
||||||
// so a sentence mentioning "B0" or "C0" should round-trip unchanged when there's nothing to replace.
|
// strings, but this case is cheap to keep as a guard.
|
||||||
val text = "Pricing tier B0 vs C0 vs A0 — see https://docs.example.com/tiers"
|
val text = "Pricing tier B0 vs C0 vs A0 — see https://docs.example.com/tiers"
|
||||||
val dict = TranslationDictionary.build(text)
|
val dict = TranslationDictionary.build(text)
|
||||||
val encoded = TranslationDictionary.encode(text, dict)
|
val encoded = TranslationDictionary.encode(text, dict)
|
||||||
@@ -280,4 +281,46 @@ class TranslationDictionaryTest {
|
|||||||
assertFalse(encoded.contains("nostr:npub126ntw5mnermmj0znhjhgdk8lh2af72sm8qfzq48umdlnhaj9kuns3le9ll"))
|
assertFalse(encoded.contains("nostr:npub126ntw5mnermmj0znhjhgdk8lh2af72sm8qfzq48umdlnhaj9kuns3le9ll"))
|
||||||
assertFalse(encoded.contains("nostr:npub1getal6ykt05fsz5nqu4uld09nfj3y3qxmv8crys4aeut53unfvlqr80nfm"))
|
assertFalse(encoded.contains("nostr:npub1getal6ykt05fsz5nqu4uld09nfj3y3qxmv8crys4aeut53unfvlqr80nfm"))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `build allocates indices above any user-supplied placeholders`() {
|
||||||
|
// User's note contains literal {0} and {3} (e.g. a code snippet). Our dictionary must not
|
||||||
|
// reuse those indices, or decode would clobber the user's own text.
|
||||||
|
val text = "Code is printf(\"%s\", arg{0}) or fmt(\"%s\", arg{3}); see https://docs.example.com"
|
||||||
|
val dict = TranslationDictionary.build(text)
|
||||||
|
assertTrue("dict should include the URL", dict.containsValue("https://docs.example.com"))
|
||||||
|
for (token in dict.keys) {
|
||||||
|
assertFalse("$token collides with user-supplied {0}", token == "{0}")
|
||||||
|
assertFalse("$token collides with user-supplied {3}", token == "{3}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `build is resilient to adversarial user placeholders beyond the table limit`() {
|
||||||
|
// Adversarial note carries `{99999}` — well past PLACEHOLDER_LIMIT. If we naively started
|
||||||
|
// our counter at max+1, the very next call to placeholder() would throw and translation
|
||||||
|
// would fail entirely. Counter must stay within the allowed range.
|
||||||
|
val text = "weird {99999} placeholder; see https://docs.example.com"
|
||||||
|
val dict = TranslationDictionary.build(text)
|
||||||
|
assertTrue("dict should include the URL", dict.containsValue("https://docs.example.com"))
|
||||||
|
// {99999} is out of range and never enters our dict, so it survives encode/decode untouched.
|
||||||
|
val encoded = TranslationDictionary.encode(text, dict)
|
||||||
|
assertTrue("user's {99999} must survive encode", encoded.contains("{99999}"))
|
||||||
|
val decoded = TranslationDictionary.decode(encoded, dict)
|
||||||
|
assertEquals(text, decoded)
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
fun `round-trip preserves user-supplied placeholders in source text`() {
|
||||||
|
// The user wrote {0} and {3}; after encode/decode they must round-trip intact alongside our
|
||||||
|
// own placeholder for the URL.
|
||||||
|
val text = "Code: printf(\"%s\", arg{0}) and fmt(\"%s\", arg{3}); see https://docs.example.com"
|
||||||
|
val dict = TranslationDictionary.build(text)
|
||||||
|
val encoded = TranslationDictionary.encode(text, dict)
|
||||||
|
// Encoded form still contains the user's {0} and {3} verbatim — we never touched them.
|
||||||
|
assertTrue(encoded.contains("arg{0}"))
|
||||||
|
assertTrue(encoded.contains("arg{3}"))
|
||||||
|
val decoded = TranslationDictionary.decode(encoded, dict)
|
||||||
|
assertEquals(text, decoded)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user