This is an automated email from the ASF dual-hosted git repository. garydgregory pushed a commit to branch master in repository https://gitbox.apache.org/repos/asf/commons-lang.git
commit a9104066058a3f98b25266d66df45f9b10b9eaee Author: Gary Gregory <[email protected]> AuthorDate: Wed Oct 7 15:47:23 2026 -0400 [LANG-1655] Preserve Korean text in stripAccents (#1799) Add stripAccents regression tests for Unicode recomposition Cover Japanese and Bengali recomposition, modern and compatibility Hangul Jamo, and mixed Korean and accented Latin text. --- .../apache/commons/lang3/StringUtilsStripTest.java | 31 ++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/src/test/java/org/apache/commons/lang3/StringUtilsStripTest.java b/src/test/java/org/apache/commons/lang3/StringUtilsStripTest.java index 5cb02323c..4eb73f686 100644 --- a/src/test/java/org/apache/commons/lang3/StringUtilsStripTest.java +++ b/src/test/java/org/apache/commons/lang3/StringUtilsStripTest.java @@ -54,11 +54,26 @@ void testStripAccents() { "Failed to handle non-accented text"); } + @Test + void testStripAccentsBengali() { + // Bengali vowel sign O, both precomposed and decomposed. + assertEquals("\u09CB", StringUtils.stripAccents("\u09CB")); + assertEquals("\u09CB", StringUtils.stripAccents("\u09C7\u09BE")); + } + @Test void testStripAccentsIWithBar() { assertEquals("I i I i I", StringUtils.stripAccents("\u0197 \u0268 \u1D7B \u1DA4 \u1DA7")); } + @Test + void testStripAccentsJapanese() { + // Katakana GA retains its dakuten in precomposed, decomposed, and halfwidth forms. + assertEquals("\u30AC", StringUtils.stripAccents("\u30AC")); + assertEquals("\u30AC", StringUtils.stripAccents("\u30AB\u3099")); + assertEquals("\u30AC", StringUtils.stripAccents("\uFF76\uFF9E")); + } + @Test void testStripAccentsKorean() { // LANG-1655 @@ -66,6 +81,22 @@ void testStripAccentsKorean() { assertEquals(input, StringUtils.stripAccents(input), "Failed to handle Korean text"); } + @Test + void testStripAccentsKoreanJamo() { + // Compose modern Jamo with and without a trailing consonant. + assertEquals("\uAC00", StringUtils.stripAccents("\u1100\u1161")); + assertEquals("\uAC01", StringUtils.stripAccents("\u1100\u1161\u11A8")); + // Compatibility Jamo are still folded before composition. + assertEquals("\uAC00", StringUtils.stripAccents("\u3131\u314F")); + } + + @Test + void testStripAccentsKoreanWithLatinAccents() { + final String expected = "\uAC00 \uAC01 cafe deja vu"; + assertEquals(expected, StringUtils.stripAccents("\uAC00 \uAC01 caf\u00E9 d\u00E9j\u00E0 vu")); + assertEquals(expected, StringUtils.stripAccents("\u1100\u1161 \u1100\u1161\u11A8 cafe\u0301 de\u0301ja\u0300 vu")); + } + /** * Decomposes ligatures and digraphs per the KD column in the <a href = "https://www.unicode.org/charts/normalization/">Unicode Normalization Chart.</a> */
