This is an automated email from the ASF dual-hosted git repository.

davsclaus pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/camel.git


The following commit(s) were added to refs/heads/main by this push:
     new ac7c5c76729b CAMEL-25214: camel-hl7 - use the right charsets for 
MSH-18 8859/6..9, 8859/15 and GB 18030-2000 (#27163)
ac7c5c76729b is described below

commit ac7c5c76729b5cada2ebdcbdcd4d806de758800c
Author: allthingssecurity <[email protected]>
AuthorDate: Thu Oct 1 12:24:25 2026 +0530

    CAMEL-25214: camel-hl7 - use the right charsets for MSH-18 8859/6..9, 
8859/15 and GB 18030-2000 (#27163)
    
    HL7Charset listed ISO-8859-6 .. ISO-8859-9 under the HL7 name 8859/1
    (copy and paste), so MSH-18 values 8859/6, 8859/7, 8859/8 and 8859/9 were
    not found and the data format fell back to the exchange charset (UTF-8 by
    default), which corrupts Arabic, Greek, Hebrew and Turkish text. 8859/15
    was missing, and GB 18030-2000 mapped to an empty Java charset name, so
    unmarshal and marshal failed with UnsupportedEncodingException.
    
    The table now matches HL7 table 0211 as mapped by HAPI (HL7Charsets) and
    camel-mllp (MllpProtocolConstants.MSH18_VALUES). The Japanese and CNS
    entries are left as they are.
    
    Co-Authored-By: Claude Opus 5.5 <[email protected]>
---
 .../org/apache/camel/component/hl7/HL7Charset.java |  11 +-
 .../hl7/HL7DataFormatMsh18CharsetTest.java         | 142 +++++++++++++++++++++
 .../ROOT/pages/camel-4x-upgrade-guide-4_23.adoc    |   9 ++
 3 files changed, 157 insertions(+), 5 deletions(-)

diff --git 
a/components/camel-hl7/src/main/java/org/apache/camel/component/hl7/HL7Charset.java
 
b/components/camel-hl7/src/main/java/org/apache/camel/component/hl7/HL7Charset.java
index 165d0a4249ab..eace5e63724f 100644
--- 
a/components/camel-hl7/src/main/java/org/apache/camel/component/hl7/HL7Charset.java
+++ 
b/components/camel-hl7/src/main/java/org/apache/camel/component/hl7/HL7Charset.java
@@ -36,14 +36,15 @@ public enum HL7Charset {
     ISO_8859_3("8859/3", "ISO-8859-3"),
     ISO_8859_4("8859/4", "ISO-8859-4"),
     ISO_8859_5("8859/5", "ISO-8859-5"),
-    ISO_8859_6("8859/1", "ISO-8859-6"),
-    ISO_8859_7("8859/1", "ISO-8859-7"),
-    ISO_8859_8("8859/1", "ISO-8859-8"),
-    ISO_8859_9("8859/1", "ISO-8859-9"),
+    ISO_8859_6("8859/6", "ISO-8859-6"),
+    ISO_8859_7("8859/7", "ISO-8859-7"),
+    ISO_8859_8("8859/8", "ISO-8859-8"),
+    ISO_8859_9("8859/9", "ISO-8859-9"),
+    ISO_8859_15("8859/15", "ISO-8859-15"),
     ASCII("ASCII", "US-ASCII"),
     BIG_5("BIG-5", "Big5"),
     CNS("CNS 11643-1992", "ISO-2022-CN"),
-    GB_1830_2000("GB 18030-2000", ""),
+    GB_1830_2000("GB 18030-2000", "GB18030"),
     ISO_IR14("ISO IR14", "ISO-2022-JP"),
     ISO_IR159("ISO IR159", "EUC-JP"),
     ISO_IR87("ISO IR87", "EUC-JP"),
diff --git 
a/components/camel-hl7/src/test/java/org/apache/camel/component/hl7/HL7DataFormatMsh18CharsetTest.java
 
b/components/camel-hl7/src/test/java/org/apache/camel/component/hl7/HL7DataFormatMsh18CharsetTest.java
new file mode 100644
index 000000000000..9aa074516ba9
--- /dev/null
+++ 
b/components/camel-hl7/src/test/java/org/apache/camel/component/hl7/HL7DataFormatMsh18CharsetTest.java
@@ -0,0 +1,142 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *      http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.camel.component.hl7;
+
+import java.io.ByteArrayInputStream;
+import java.nio.charset.Charset;
+
+import ca.uhn.hl7v2.DefaultHapiContext;
+import ca.uhn.hl7v2.HapiContext;
+import ca.uhn.hl7v2.model.Message;
+import ca.uhn.hl7v2.model.v24.message.ADR_A19;
+import ca.uhn.hl7v2.util.Terser;
+import org.apache.camel.Exchange;
+import org.apache.camel.builder.RouteBuilder;
+import org.apache.camel.component.mock.MockEndpoint;
+import org.apache.camel.test.junit6.CamelTestSupport;
+import org.junit.jupiter.api.Test;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+
+/**
+ * The HL7 data format reads and writes a message in the character set named 
in MSH-18 (HL7 table 0211).
+ */
+public class HL7DataFormatMsh18CharsetTest extends CamelTestSupport {
+
+    private static final String GREEK = "Παπαδόπουλος";
+    private static final String TURKISH = "Şahİn Öztürk ğı";
+    private static final String HEBREW = "כהן";
+    private static final String ARABIC = "محمد";
+    private static final String CHINESE = "张伟";
+    private static final String LATIN9 = "€ ŠšŽž";
+    private static final String CYRILLIC = "Иванов";
+
+    @Test
+    public void testUnmarshalIso88596() throws Exception {
+        assertUnmarshal("8859/6", "ISO-8859-6", ARABIC);
+    }
+
+    @Test
+    public void testUnmarshalIso88597() throws Exception {
+        assertUnmarshal("8859/7", "ISO-8859-7", GREEK);
+    }
+
+    @Test
+    public void testUnmarshalIso88598() throws Exception {
+        assertUnmarshal("8859/8", "ISO-8859-8", HEBREW);
+    }
+
+    @Test
+    public void testUnmarshalIso88599() throws Exception {
+        assertUnmarshal("8859/9", "ISO-8859-9", TURKISH);
+    }
+
+    @Test
+    public void testUnmarshalIso885915() throws Exception {
+        assertUnmarshal("8859/15", "ISO-8859-15", LATIN9);
+    }
+
+    @Test
+    public void testUnmarshalGb18030() throws Exception {
+        assertUnmarshal("GB 18030-2000", "GB18030", CHINESE);
+    }
+
+    @Test
+    public void testUnmarshalIso88595() throws Exception {
+        // control: this entry of the table is right
+        assertUnmarshal("8859/5", "ISO-8859-5", CYRILLIC);
+    }
+
+    @Test
+    public void testMarshalIso88597() throws Exception {
+        assertMarshal("8859/7", "ISO-8859-7", GREEK);
+    }
+
+    @Test
+    public void testMarshalGb18030() throws Exception {
+        assertMarshal("GB 18030-2000", "GB18030", CHINESE);
+    }
+
+    private void assertUnmarshal(String msh18, String javaCharset, String 
name) throws Exception {
+        MockEndpoint mock = getMockEndpoint("mock:unmarshal");
+        mock.expectedMessageCount(1);
+
+        String hl7 = 
"MSH|^~\\&|MYSENDER|MYSENDERAPP|MYCLIENT|MYCLIENTAPP|200612211200||QRY^A19|1234|P|2.4||||||"
 + msh18
+                     + "\rQRD|200612211200|R|I|GetPatient|||1^RD|0101701234^" 
+ name + "|DEM||";
+        template.sendBody("direct:unmarshal", new 
ByteArrayInputStream(hl7.getBytes(Charset.forName(javaCharset))));
+
+        MockEndpoint.assertIsSatisfied(context);
+        Exchange exchange = mock.getReceivedExchanges().get(0);
+        assertEquals(javaCharset, 
exchange.getIn().getHeader(Exchange.CHARSET_NAME));
+        Message message = exchange.getIn().getBody(Message.class);
+        assertEquals(name, new Terser(message).get("QRD-8-2"));
+    }
+
+    private void assertMarshal(String msh18, String javaCharset, String name) 
throws Exception {
+        MockEndpoint mock = getMockEndpoint("mock:marshal");
+        mock.expectedMessageCount(1);
+
+        ADR_A19 adr = new ADR_A19();
+        adr.getMSH().getFieldSeparator().setValue("|");
+        adr.getMSH().getEncodingCharacters().setValue("^~\\&");
+        adr.getMSH().getMessageType().getMessageType().setValue("ADR");
+        adr.getMSH().getMessageType().getTriggerEvent().setValue("A19");
+        adr.getMSH().getVersionID().getVersionID().setValue("2.4");
+        adr.getMSH().getCharacterSet(0).setValue(msh18);
+        adr.getMSA().getAcknowledgementCode().setValue("AA");
+        adr.getMSA().getMessageControlID().setValue("123");
+        adr.getMSA().getMsa3_TextMessage().setValue(name);
+        template.sendBody("direct:marshal", adr);
+
+        MockEndpoint.assertIsSatisfied(context);
+        byte[] body = 
mock.getReceivedExchanges().get(0).getIn().getBody(byte[].class);
+        String text = new String(body, Charset.forName(javaCharset));
+        try (HapiContext hapi = new DefaultHapiContext()) {
+            assertEquals(name, new 
Terser(hapi.getGenericParser().parse(text)).get("MSA-3"));
+        }
+    }
+
+    @Override
+    protected RouteBuilder createRouteBuilder() {
+        return new RouteBuilder() {
+            public void configure() {
+                from("direct:marshal").marshal().hl7().to("mock:marshal");
+                
from("direct:unmarshal").unmarshal().hl7(false).to("mock:unmarshal");
+            }
+        };
+    }
+}
diff --git 
a/docs/user-manual/modules/ROOT/pages/camel-4x-upgrade-guide-4_23.adoc 
b/docs/user-manual/modules/ROOT/pages/camel-4x-upgrade-guide-4_23.adoc
index 5e046ad7e58c..bb817e1253fc 100644
--- a/docs/user-manual/modules/ROOT/pages/camel-4x-upgrade-guide-4_23.adoc
+++ b/docs/user-manual/modules/ROOT/pages/camel-4x-upgrade-guide-4_23.adoc
@@ -698,6 +698,15 @@ short by that count, so it now fails the record `length` 
check (unless `ignoreMi
 read with shifted fields. A system that reads these records by counting UTF-16 
chars must count code points (or
 graphemes) instead.
 
+=== camel-hl7 - character sets from MSH-18
+
+The HL7 data format now maps the MSH-18 values `8859/6`, `8859/7`, `8859/8` 
and `8859/9` to ISO-8859-6 .. ISO-8859-9,
+`8859/15` to ISO-8859-15, and `GB 18030-2000` to GB18030, as in HL7 table 0211 
(and as camel-mllp already does).
+Before, the first five fell back to the charset of the exchange (UTF-8 by 
default), and `GB 18030-2000` failed with
+`UnsupportedEncodingException`. Messages with these MSH-18 values are now read 
and written in the named charset, and
+unmarshal sets `CamelCharsetName` to it. A message whose MSH-18 names one of 
these charsets but that is actually
+encoded in another one (such as UTF-8) is now decoded in the charset it names.
+
 === Components and Language removal
 
 ==== camel-csimple, camel-csimple-joor and csimple-maven-plugin

Reply via email to