This is an automated email from the ASF dual-hosted git repository.

garydgregory pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/commons-lang.git

commit a9104066058a3f98b25266d66df45f9b10b9eaee
Author: Gary Gregory <[email protected]>
AuthorDate: Wed Oct 7 15:47:23 2026 -0400

    [LANG-1655] Preserve Korean text in stripAccents (#1799)
    
    Add stripAccents regression tests for Unicode recomposition
    
    Cover Japanese and Bengali recomposition, modern and compatibility
    Hangul Jamo, and mixed Korean and accented Latin text.
---
 .../apache/commons/lang3/StringUtilsStripTest.java | 31 ++++++++++++++++++++++
 1 file changed, 31 insertions(+)

diff --git a/src/test/java/org/apache/commons/lang3/StringUtilsStripTest.java 
b/src/test/java/org/apache/commons/lang3/StringUtilsStripTest.java
index 5cb02323c..4eb73f686 100644
--- a/src/test/java/org/apache/commons/lang3/StringUtilsStripTest.java
+++ b/src/test/java/org/apache/commons/lang3/StringUtilsStripTest.java
@@ -54,11 +54,26 @@ void testStripAccents() {
                 "Failed to handle non-accented text");
     }
 
+    @Test
+    void testStripAccentsBengali() {
+        // Bengali vowel sign O, both precomposed and decomposed.
+        assertEquals("\u09CB", StringUtils.stripAccents("\u09CB"));
+        assertEquals("\u09CB", StringUtils.stripAccents("\u09C7\u09BE"));
+    }
+
     @Test
     void testStripAccentsIWithBar() {
         assertEquals("I i I i I", StringUtils.stripAccents("\u0197 \u0268 
\u1D7B \u1DA4 \u1DA7"));
     }
 
+    @Test
+    void testStripAccentsJapanese() {
+        // Katakana GA retains its dakuten in precomposed, decomposed, and 
halfwidth forms.
+        assertEquals("\u30AC", StringUtils.stripAccents("\u30AC"));
+        assertEquals("\u30AC", StringUtils.stripAccents("\u30AB\u3099"));
+        assertEquals("\u30AC", StringUtils.stripAccents("\uFF76\uFF9E"));
+    }
+
     @Test
     void testStripAccentsKorean() {
         // LANG-1655
@@ -66,6 +81,22 @@ void testStripAccentsKorean() {
         assertEquals(input, StringUtils.stripAccents(input), "Failed to handle 
Korean text");
     }
 
+    @Test
+    void testStripAccentsKoreanJamo() {
+        // Compose modern Jamo with and without a trailing consonant.
+        assertEquals("\uAC00", StringUtils.stripAccents("\u1100\u1161"));
+        assertEquals("\uAC01", StringUtils.stripAccents("\u1100\u1161\u11A8"));
+        // Compatibility Jamo are still folded before composition.
+        assertEquals("\uAC00", StringUtils.stripAccents("\u3131\u314F"));
+    }
+
+    @Test
+    void testStripAccentsKoreanWithLatinAccents() {
+        final String expected = "\uAC00 \uAC01 cafe deja vu";
+        assertEquals(expected, StringUtils.stripAccents("\uAC00 \uAC01 
caf\u00E9 d\u00E9j\u00E0 vu"));
+        assertEquals(expected, StringUtils.stripAccents("\u1100\u1161 
\u1100\u1161\u11A8 cafe\u0301 de\u0301ja\u0300 vu"));
+    }
+
     /**
      * Decomposes ligatures and digraphs per the KD column in the <a href = 
"https://www.unicode.org/charts/normalization/";>Unicode Normalization Chart.</a>
      */

Reply via email to