This is an automated email from the ASF dual-hosted git repository.

tballison pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/tika.git


The following commit(s) were added to refs/heads/main by this push:
     new 162f939994 TIKA-4939: improve xml testing (#3270)
162f939994 is described below

commit 162f93999418ee9f3c2cade9befaf843c27d449d
Author: Tim Allison <[email protected]>
AuthorDate: Mon Sep 28 15:46:28 2026 -0400

    TIKA-4939: improve xml testing (#3270)
---
 .skills/devs/development/SKILL.md                  |   7 +
 CHANGES.txt                                        |   4 +
 tika-core/pom.xml                                  |  60 +++
 .../java/org/apache/tika/mime/MimeTypesReader.java |   3 +
 .../org/apache/tika/utils/SuppressForbidden.java   |  32 ++
 .../java/org/apache/tika/utils/XMLReaderUtils.java |  20 +
 .../apache/tika/utils/XmlSecurityContractTest.java | 340 +++++++++++++++++
 .../tika/eval/app/reports/ResultsReporter.java     |   8 +-
 .../apache/tika/eval/structure/BlockExtractor.java |   4 +-
 tika-parent/forbidden-apis-xml-factory-bans.txt    |  81 ++++
 tika-parent/pom.xml                                |   5 +
 .../geoinfo/GeographicInformationXxeTest.java      | 142 +++++++
 .../apache/tika/parser/TestXMLEntityExpansion.java |  22 --
 .../java/org/apache/tika/parser/TestXXEInXML.java  | 425 +++++++++++++++++++++
 .../java/org/apache/tika/parser/XMLTestBase.java   |  72 ++--
 .../ooxml/xwpf/ml2006/Word2006MLParser.java        |   9 +-
 .../microsoft/xml/AbstractXML2003Parser.java       |   9 +-
 .../tika/async/cli/TikaConfigAsyncWriter.java      |   2 +-
 18 files changed, 1174 insertions(+), 71 deletions(-)

diff --git a/.skills/devs/development/SKILL.md 
b/.skills/devs/development/SKILL.md
index 43ee5377a5..9a04a78968 100644
--- a/.skills/devs/development/SKILL.md
+++ b/.skills/devs/development/SKILL.md
@@ -156,6 +156,13 @@ personal configuration does not override it.
   invariant, workaround, spec quirk).  Never restate the code, narrate the
   next line, justify the change to a reviewer, or describe past states of
   the code.
+- **XML is parsed only through `XMLReaderUtils.parseSAX` / `buildDOM`** 
(forbidden-apis
+  rejects the JAXP and Commons Secure XML factories and the raw `getSAXParser` 
/
+  `getXMLReader` / `getDocumentBuilder` in main code). Never install an 
`EntityResolver`;
+  a resolver that returns null or a stream-less `InputSource` makes the parser 
fetch the
+  id itself (CVE-2025-66516). A new XML-bearing format gets a fixture in 
`TestXXEInXML`;
+  a third-party library that parses XML with its own parser gets its own 
oracle test
+  (see `GeographicInformationXxeTest`) before it ships.
 - **Input files are hostile**: bound anything derived from document content
   (loop counts, allocations, timeouts); release external processes, temp
   files, and pool slots on every failure path.
diff --git a/CHANGES.txt b/CHANGES.txt
index 4e6124d596..1f75ba7f27 100644
--- a/CHANGES.txt
+++ b/CHANGES.txt
@@ -1,5 +1,9 @@
 Release 4.2.0 - unreleased
 
+   * XmlSecurityContractTest pins XMLReaderUtils' XXE and entity-expansion 
defenses
+     forbidden-apis now rejects building JAXP or Commons Secure XML parsers 
anywhere but 
+     XMLReaderUtils (TIKA-4939).
+     
    * Retire Tika's entity expansion limit (default 20) in SAX, DOM and StAX
      parsing in favor of standard Java configuration methods (TIKA-4940).
    
diff --git a/tika-core/pom.xml b/tika-core/pom.xml
index aef0860055..e8cbf7fe87 100644
--- a/tika-core/pom.xml
+++ b/tika-core/pom.xml
@@ -125,6 +125,10 @@
     </dependency>
   </dependencies>
 
+  <properties>
+    <xerces.test.version>2.12.2</xerces.test.version>
+  </properties>
+
   <build>
     <plugins>
       <plugin>
@@ -257,6 +261,62 @@
           </execution>
         </executions>
       </plugin> -->
+      <!-- XmlSecurityContractTest runs once per JAXP provider Tika verifies: 
the default
+           surefire execution on the JDK's Xerces, and this one with 
standalone Xerces on the
+           classpath. The jars are copied, not declared, so nothing else in 
tika-core sees them
+           (the enforcer bans them as dependencies). -->
+      <plugin>
+        <groupId>org.apache.maven.plugins</groupId>
+        <artifactId>maven-dependency-plugin</artifactId>
+        <executions>
+          <execution>
+            <id>copy-xml-providers</id>
+            <phase>generate-test-resources</phase>
+            <goals>
+              <goal>copy</goal>
+            </goals>
+            <configuration>
+              <artifactItems>
+                <artifactItem>
+                  <groupId>xerces</groupId>
+                  <artifactId>xercesImpl</artifactId>
+                  <version>${xerces.test.version}</version>
+                  <destFileName>xercesImpl.jar</destFileName>
+                </artifactItem>
+                <artifactItem>
+                  <groupId>xml-apis</groupId>
+                  <artifactId>xml-apis</artifactId>
+                  <version>1.4.01</version>
+                  <destFileName>xml-apis.jar</destFileName>
+                </artifactItem>
+              </artifactItems>
+              
<outputDirectory>${project.build.directory}/xml-providers</outputDirectory>
+            </configuration>
+          </execution>
+        </executions>
+      </plugin>
+      <plugin>
+        <groupId>org.apache.maven.plugins</groupId>
+        <artifactId>maven-surefire-plugin</artifactId>
+        <executions>
+          <execution>
+            <id>xml-provider-xerces</id>
+            <goals>
+              <goal>test</goal>
+            </goals>
+            <configuration>
+              <test>XmlSecurityContractTest</test>
+              <additionalClasspathElements>
+                
<additionalClasspathElement>${project.build.directory}/xml-providers/xercesImpl.jar</additionalClasspathElement>
+                
<additionalClasspathElement>${project.build.directory}/xml-providers/xml-apis.jar</additionalClasspathElement>
+              </additionalClasspathElements>
+              <systemPropertyVariables>
+                <tika.test.xml.provider>xerces</tika.test.xml.provider>
+              </systemPropertyVariables>
+            </configuration>
+          </execution>
+        </executions>
+      </plugin>
       <plugin>
         <artifactId>maven-failsafe-plugin</artifactId>
         <version>${maven.failsafe.version}</version>
diff --git a/tika-core/src/main/java/org/apache/tika/mime/MimeTypesReader.java 
b/tika-core/src/main/java/org/apache/tika/mime/MimeTypesReader.java
index c1bc031c74..a350939f32 100644
--- a/tika-core/src/main/java/org/apache/tika/mime/MimeTypesReader.java
+++ b/tika-core/src/main/java/org/apache/tika/mime/MimeTypesReader.java
@@ -45,6 +45,7 @@ import org.xml.sax.SAXException;
 import org.xml.sax.helpers.DefaultHandler;
 
 import org.apache.tika.exception.TikaException;
+import org.apache.tika.utils.SuppressForbidden;
 import org.apache.tika.utils.XMLReaderUtils;
 
 /**
@@ -210,6 +211,8 @@ public class MimeTypesReader extends DefaultHandler 
implements MimeTypesReaderMe
         }
     }
 
+    //non-namespace parser for the trusted type registry only
+    @SuppressForbidden
     private static SAXParser newSAXParser() throws TikaException {
         SAXParserFactory factory = SecureSAXParserFactory.newInstance();
         try {
diff --git 
a/tika-core/src/main/java/org/apache/tika/utils/SuppressForbidden.java 
b/tika-core/src/main/java/org/apache/tika/utils/SuppressForbidden.java
new file mode 100644
index 0000000000..52b275d0d2
--- /dev/null
+++ b/tika-core/src/main/java/org/apache/tika/utils/SuppressForbidden.java
@@ -0,0 +1,32 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.utils;
+
+import java.lang.annotation.ElementType;
+import java.lang.annotation.Retention;
+import java.lang.annotation.RetentionPolicy;
+import java.lang.annotation.Target;
+
+/**
+ * Exempts a class or member from the forbidden-apis bans in tika-parent. Use 
only where
+ * the banned call is the chokepoint itself, such as {@link XMLReaderUtils} 
building the
+ * XML parsers every other caller must obtain from it.
+ */
+@Retention(RetentionPolicy.CLASS)
+@Target({ElementType.TYPE, ElementType.METHOD, ElementType.CONSTRUCTOR, 
ElementType.FIELD})
+public @interface SuppressForbidden {
+}
diff --git a/tika-core/src/main/java/org/apache/tika/utils/XMLReaderUtils.java 
b/tika-core/src/main/java/org/apache/tika/utils/XMLReaderUtils.java
index da1119f088..5570fbf8d1 100644
--- a/tika-core/src/main/java/org/apache/tika/utils/XMLReaderUtils.java
+++ b/tika-core/src/main/java/org/apache/tika/utils/XMLReaderUtils.java
@@ -66,6 +66,7 @@ import org.apache.tika.sax.OfflineContentHandler;
 /**
  * Utility functions for reading XML.
  */
+@SuppressForbidden
 public class XMLReaderUtils implements Serializable {
 
     /**
@@ -178,6 +179,8 @@ public class XMLReaderUtils implements Serializable {
      * @see #getSAXParserFactory()
      * @since Apache Tika 0.8
      */
+    // a raw parser honors a caller resolver that answers with a bare system 
id; in-tree
+    // code parses through parseSAX, which shadows the caller's resolver 
(TIKA-4939)
     public static SAXParser getSAXParser() throws TikaException {
         try {
             return getSAXParserFactory().newSAXParser();
@@ -239,6 +242,15 @@ public class XMLReaderUtils implements Serializable {
      * @return DOM Builder
      * @since Apache Tika 1.13
      */
+    public static Document newDocument() throws TikaException {
+        return getDocumentBuilder().newDocument();
+    }
+
+    /**
+     * Returns a builder that accepts a caller-supplied {@link 
org.xml.sax.EntityResolver},
+     * which a parser will honor even when it answers with a bare system id. 
Parse
+     * untrusted XML with {@link #buildDOM(InputStream, ParseContext)} instead.
+     */
     public static DocumentBuilder getDocumentBuilder() throws TikaException {
         try {
             DocumentBuilderFactory documentBuilderFactory = 
getDocumentBuilderFactory();
@@ -361,6 +373,8 @@ public class XMLReaderUtils implements Serializable {
             }
         }
 
+        //a supplied builder never brings its own resolver along
+        builder.setEntityResolver(IGNORING_SAX_ENTITY_RESOLVER);
         try {
             return builder.parse(is);
         } finally {
@@ -397,6 +411,8 @@ public class XMLReaderUtils implements Serializable {
             }
         }
 
+        //a supplied builder never brings its own resolver along
+        builder.setEntityResolver(IGNORING_SAX_ENTITY_RESOLVER);
         try {
             return builder.parse(new InputSource(reader));
         } finally {
@@ -445,6 +461,8 @@ public class XMLReaderUtils implements Serializable {
             }
         }
 
+        //a supplied builder never brings its own resolver along
+        builder.setEntityResolver(IGNORING_SAX_ENTITY_RESOLVER);
         try {
             return builder.parse(uriString);
         } finally {
@@ -476,6 +494,8 @@ public class XMLReaderUtils implements Serializable {
             }
         }
 
+        //a supplied builder never brings its own resolver along
+        builder.setEntityResolver(IGNORING_SAX_ENTITY_RESOLVER);
         try {
             return builder.parse(is);
         } finally {
diff --git 
a/tika-core/src/test/java/org/apache/tika/utils/XmlSecurityContractTest.java 
b/tika-core/src/test/java/org/apache/tika/utils/XmlSecurityContractTest.java
new file mode 100644
index 0000000000..9a849da9af
--- /dev/null
+++ b/tika-core/src/test/java/org/apache/tika/utils/XmlSecurityContractTest.java
@@ -0,0 +1,340 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.utils;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.junit.jupiter.api.Assertions.fail;
+import static org.junit.jupiter.api.Assumptions.assumeFalse;
+
+import java.io.ByteArrayInputStream;
+import java.io.FileNotFoundException;
+import java.io.IOException;
+import java.io.StringReader;
+import java.net.ConnectException;
+import java.net.NoRouteToHostException;
+import java.net.UnknownHostException;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
+import javax.xml.parsers.DocumentBuilder;
+import javax.xml.parsers.DocumentBuilderFactory;
+import javax.xml.parsers.SAXParser;
+import javax.xml.parsers.SAXParserFactory;
+import javax.xml.transform.Transformer;
+import javax.xml.transform.TransformerFactory;
+import javax.xml.transform.dom.DOMResult;
+import javax.xml.transform.stream.StreamSource;
+
+import org.junit.jupiter.api.BeforeAll;
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.Timeout;
+import org.junit.jupiter.api.io.TempDir;
+import org.junit.jupiter.params.ParameterizedTest;
+import org.junit.jupiter.params.provider.ValueSource;
+import org.w3c.dom.Document;
+import org.xml.sax.InputSource;
+import org.xml.sax.SAXException;
+import org.xml.sax.helpers.DefaultHandler;
+
+import org.apache.tika.exception.TikaException;
+import org.apache.tika.parser.ParseContext;
+
+/**
+ * The XML security contract of {@link XMLReaderUtils}: no external resource 
is ever
+ * fetched, and entity expansion is bounded. Runs once per JAXP provider Tika 
verifies;
+ * the tika-core pom runs it again with standalone Xerces on the classpath and
+ * {@code tika.test.xml.provider=xerces}. A wrong provider fails the run 
rather than
+ * silently testing the wrong parser.
+ */
+public class XmlSecurityContractTest {
+
+    private static final String PROVIDER_PROPERTY = "tika.test.xml.provider";
+    private static final String SECRET = "SECRET_CONTENT_7f3a";
+    // more than any bounded parse should hand a handler
+    private static final long MAX_EXPANDED_CHARS = 50_000_000L;
+
+    private static final String UNROUTABLE = "http://127.234.172.38:7845/bar";;
+    private static final String CLOSED_PORT = "http://127.0.0.1:9/bar";;
+
+    private static final String BILLION_LAUGHS = "<?xml 
version=\"1.0\"?>\n<!DOCTYPE lolz [\n" +
+            " <!ENTITY lol \"lol\">\n <!ELEMENT lolz (#PCDATA)>\n" +
+            " <!ENTITY lol1 
\"&lol;&lol;&lol;&lol;&lol;&lol;&lol;&lol;&lol;&lol;\">\n" +
+            " <!ENTITY lol2 
\"&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;\">\n" +
+            " <!ENTITY lol3 
\"&lol2;&lol2;&lol2;&lol2;&lol2;&lol2;&lol2;&lol2;&lol2;&lol2;\">\n" +
+            " <!ENTITY lol4 
\"&lol3;&lol3;&lol3;&lol3;&lol3;&lol3;&lol3;&lol3;&lol3;&lol3;\">\n" +
+            " <!ENTITY lol5 
\"&lol4;&lol4;&lol4;&lol4;&lol4;&lol4;&lol4;&lol4;&lol4;&lol4;\">\n" +
+            " <!ENTITY lol6 
\"&lol5;&lol5;&lol5;&lol5;&lol5;&lol5;&lol5;&lol5;&lol5;&lol5;\">\n" +
+            " <!ENTITY lol7 
\"&lol6;&lol6;&lol6;&lol6;&lol6;&lol6;&lol6;&lol6;&lol6;&lol6;\">\n" +
+            " <!ENTITY lol8 
\"&lol7;&lol7;&lol7;&lol7;&lol7;&lol7;&lol7;&lol7;&lol7;&lol7;\">\n" +
+            " <!ENTITY lol9 
\"&lol8;&lol8;&lol8;&lol8;&lol8;&lol8;&lol8;&lol8;&lol8;&lol8;\">\n" +
+            "]>\n<lolz>&lol9;</lolz>";
+
+    // one 1 MB entity referenced 100k times: a size bomb rather than a depth 
bomb
+    private static final String SIZE_BOMB = "<?xml 
version=\"1.0\"?>\n<!DOCTYPE kaboom [\n" +
+            "  <!ENTITY a \"" + "a".repeat(1_000_000) + "\">]><kaboom>" +
+            "&a;".repeat(100_000) + "</kaboom>";
+
+    private static final String PROVIDER = 
System.getProperty(PROVIDER_PROPERTY, "jdk");
+
+    @TempDir
+    static Path dir;
+    static String secretUri;
+    static String leakDtdUri;
+
+    @BeforeAll
+    public static void setUp() throws IOException {
+        String expectedPackage;
+        switch (PROVIDER) {
+            case "jdk":
+                expectedPackage = "com.sun.org.apache.xerces.internal.jaxp.";
+                break;
+            case "xerces":
+                expectedPackage = "org.apache.xerces.jaxp.";
+                break;
+            default:
+                throw new IllegalArgumentException(PROVIDER_PROPERTY + "=" + 
PROVIDER);
+        }
+        // the library wraps these; check the lookup it performs, not the 
wrapper
+        String sax = SAXParserFactory.newInstance().getClass().getName();
+        String dom = DocumentBuilderFactory.newInstance().getClass().getName();
+        assertTrue(sax.startsWith(expectedPackage), PROVIDER + " expected, SAX 
provider is " + sax);
+        assertTrue(dom.startsWith(expectedPackage), PROVIDER + " expected, DOM 
provider is " + dom);
+
+        Path secret = dir.resolve("secret.txt");
+        Files.writeString(secret, SECRET, StandardCharsets.UTF_8);
+        secretUri = secret.toUri().toString();
+        Path leakDtd = dir.resolve("leak.dtd");
+        Files.writeString(leakDtd, "<!ENTITY leak \"" + SECRET + "\">", 
StandardCharsets.UTF_8);
+        leakDtdUri = leakDtd.toUri().toString();
+    }
+
+    private static List<String> xxePayloads() {
+        List<String> xmls = new ArrayList<>();
+        // external general entity: local file, unroutable host, closed port
+        xmls.add("<!DOCTYPE r [<!ENTITY x SYSTEM \"" + secretUri + 
"\">]><r>&x;</r>");
+        xmls.add("<!DOCTYPE r [<!ENTITY x SYSTEM \"" + UNROUTABLE + 
"\">]><r>&x;</r>");
+        xmls.add("<!DOCTYPE r [<!ENTITY x SYSTEM \"" + CLOSED_PORT + 
"\">]><r>&x;</r>");
+        // external DTD subset declaring the entity
+        xmls.add("<!DOCTYPE r SYSTEM \"" + leakDtdUri + "\"><r>&leak;</r>");
+        xmls.add("<?xml version=\"1.0\" standalone=\"no\"?><!DOCTYPE r SYSTEM 
\"tutorials.dtd\"><r/>");
+        xmls.add("<?xml version=\"1.0\" standalone=\"no\"?><!DOCTYPE r SYSTEM 
\"" + UNROUTABLE +
+                "\"><r/>");
+        // XInclude, in case a provider enables it by default
+        xmls.add("<r xmlns:xi=\"http://www.w3.org/2001/XInclude\";><xi:include 
href=\"" + secretUri +
+                "\" parse=\"text\"/></r>");
+        xmls.add("<r xmlns:xi=\"http://www.w3.org/2001/XInclude\";><xi:include 
href=\"" + UNROUTABLE +
+                "\"/></r>");
+        // external parameter entity pulling in the declaration
+        xmls.add("<!DOCTYPE r [<!ENTITY % p SYSTEM \"" + leakDtdUri + 
"\">%p;]><r>&leak;</r>");
+        xmls.add("<!DOCTYPE r [<!ENTITY % p SYSTEM 
\"file:///usr/local/app/schema.dtd\">%p;]><r/>");
+        return xmls;
+    }
+
+    @ParameterizedTest
+    @ValueSource(strings = {"sax", "dom", "transformer", "saxTransformer"})
+    public void testNoExternalAccess(String path) throws Exception {
+        for (String xml : xxePayloads()) {
+            String text;
+            try {
+                text = parse(path, xml);
+            } catch (Exception e) {
+                assertNotAFetch(xml, e);
+                continue;
+            }
+            assertFalse(text.contains(SECRET), path + " leaked external 
content for " + xml);
+        }
+    }
+
+    @ParameterizedTest
+    @ValueSource(strings = {"sax", "dom"})
+    @Timeout(120)
+    public void testExpansionCountBounded(String path) throws Exception {
+        assertRejected(path, BILLION_LAUGHS);
+    }
+
+    @ParameterizedTest
+    @ValueSource(strings = {"sax", "dom"})
+    @Timeout(120)
+    public void testEntitySizeBounded(String path) throws Exception {
+        // standalone Xerces 2.12 caps the expansion count but not the 
expanded size: the
+        // size bomb runs to completion there. Known gap; Tika verifies the 
JDK only.
+        assumeFalse("xerces".equals(PROVIDER), "standalone Xerces has no 
entity size limit");
+        assertRejected(path, SIZE_BOMB);
+    }
+
+    // the caller's resolver is never consulted on the parseSAX path, so a 
resolver that
+    // answers with a bare system id (which a raw parser would fetch) changes 
nothing
+    @Test
+    public void testCallerResolverShadowed() throws Exception {
+        DefaultHandler systemIdOnly = new DefaultHandler() {
+            @Override
+            public InputSource resolveEntity(String publicId, String systemId) 
{
+                return new InputSource(systemId);
+            }
+        };
+        for (String xml : xxePayloads()) {
+            try {
+                XMLReaderUtils.parseSAX(new 
ByteArrayInputStream(xml.getBytes(StandardCharsets.UTF_8)),
+                        systemIdOnly, new ParseContext());
+            } catch (Exception e) {
+                assertNotAFetch(xml, e);
+            }
+        }
+    }
+
+    // a parser supplied through the ParseContext never brings its own 
resolver along:
+    // these are raw, unsecured JDK parsers with a resolver that fetches
+    @ParameterizedTest
+    @ValueSource(strings = {"sax", "dom"})
+    public void testSuppliedParserStillOffline(String path) throws Exception {
+        ParseContext context = new ParseContext();
+        if ("sax".equals(path)) {
+            context.set(SAXParser.class, 
SAXParserFactory.newInstance().newSAXParser());
+        } else {
+            DocumentBuilder builder = 
DocumentBuilderFactory.newInstance().newDocumentBuilder();
+            builder.setEntityResolver((publicId, systemId) -> new 
InputSource(systemId));
+            context.set(DocumentBuilder.class, builder);
+        }
+        for (String xml : xxePayloads()) {
+            byte[] bytes = xml.getBytes(StandardCharsets.UTF_8);
+            String text;
+            try {
+                if ("sax".equals(path)) {
+                    BoundedTextHandler handler = new BoundedTextHandler();
+                    XMLReaderUtils.parseSAX(new ByteArrayInputStream(bytes), 
handler, context);
+                    text = handler.text();
+                } else {
+                    text = XMLReaderUtils.buildDOM(new 
ByteArrayInputStream(bytes), context)
+                            .getDocumentElement().getTextContent();
+                }
+            } catch (Exception e) {
+                assertNotAFetch(xml, e);
+                continue;
+            }
+            assertFalse(text.contains(SECRET), path + " leaked external 
content for " + xml);
+        }
+    }
+
+    @Test
+    @Timeout(120)
+    public void testLimitSurvivesPoolReuse() throws Exception {
+        for (int i = 0; i < XMLReaderUtils.getPoolSize() * 2 + 1; i++) {
+            assertRejected("sax", BILLION_LAUGHS);
+        }
+    }
+
+    private void assertRejected(String path, String xml) throws Exception {
+        String text;
+        try {
+            text = parse(path, xml);
+        } catch (SAXException | TikaException e) {
+            assertTrue(isLimitMessage(e), path + " failed without a limit 
message: " + e);
+            return;
+        }
+        // DOM with expandEntityReferences=false may hand back an empty tree 
instead
+        assertEquals(0, text.trim().length(), path + " expanded the bomb");
+    }
+
+    private static boolean isLimitMessage(Exception e) {
+        Throwable t = e;
+        while (t != null) {
+            String msg = t.getMessage();
+            if (msg != null) {
+                String m = msg.toLowerCase(Locale.ROOT);
+                if (m.contains("jaxp0001000") || m.contains("entity 
expansion") ||
+                        m.contains("entity expansions") || 
m.contains("entitysizelimit") ||
+                        (m.contains("entit") && m.contains("limit"))) {
+                    return true;
+                }
+            }
+            t = t.getCause();
+        }
+        return false;
+    }
+
+    private static void assertNotAFetch(String xml, Exception e) {
+        Throwable t = e;
+        while (t != null) {
+            if (t instanceof ConnectException || t instanceof 
UnknownHostException ||
+                    t instanceof NoRouteToHostException || t instanceof 
FileNotFoundException) {
+                fail("parser tried to fetch an external resource for " + xml, 
e);
+            }
+            String msg = t.getMessage();
+            if (msg != null && (msg.contains("Connection refused") || 
msg.contains("No such file") ||
+                    msg.contains("Exception scanning External"))) {
+                fail("parser tried to fetch an external resource for " + xml, 
e);
+            }
+            t = t.getCause();
+        }
+    }
+
+    private static String parse(String path, String xml) throws Exception {
+        switch (path) {
+            case "sax": {
+                BoundedTextHandler handler = new BoundedTextHandler();
+                XMLReaderUtils.parseSAX(new 
ByteArrayInputStream(xml.getBytes(StandardCharsets.UTF_8)),
+                        handler, new ParseContext());
+                return handler.text();
+            }
+            case "dom": {
+                Document doc = XMLReaderUtils.buildDOM(
+                        new 
ByteArrayInputStream(xml.getBytes(StandardCharsets.UTF_8)),
+                        new ParseContext());
+                return doc.getDocumentElement().getTextContent();
+            }
+            case "transformer":
+            case "saxTransformer": {
+                TransformerFactory factory = "sax".equals(path.substring(0, 
3)) ?
+                        XMLReaderUtils.getSAXTransformerFactory() :
+                        XMLReaderUtils.getTransformerFactory();
+                Transformer transformer = factory.newTransformer();
+                DOMResult result = new DOMResult();
+                transformer.transform(new StreamSource(new StringReader(xml)), 
result);
+                return ((Document) 
result.getNode()).getDocumentElement().getTextContent();
+            }
+            default:
+                throw new IllegalArgumentException(path);
+        }
+    }
+
+    // keeps a prefix for assertions, counts everything, and stops a runaway 
expansion
+    private static class BoundedTextHandler extends DefaultHandler {
+        private final StringBuilder prefix = new StringBuilder();
+        private long count = 0;
+
+        @Override
+        public void characters(char[] ch, int start, int length) throws 
SAXException {
+            count += length;
+            if (count > MAX_EXPANDED_CHARS) {
+                throw new SAXException("harness: more than " + 
MAX_EXPANDED_CHARS +
+                        " characters expanded");
+            }
+            if (prefix.length() < 10_000) {
+                prefix.append(ch, start, Math.min(length, 10_000 - 
prefix.length()));
+            }
+        }
+
+        String text() {
+            return prefix.toString();
+        }
+    }
+}
diff --git 
a/tika-eval/tika-eval-app/src/main/java/org/apache/tika/eval/app/reports/ResultsReporter.java
 
b/tika-eval/tika-eval-app/src/main/java/org/apache/tika/eval/app/reports/ResultsReporter.java
index b7afe35e60..d01d6ff024 100644
--- 
a/tika-eval/tika-eval-app/src/main/java/org/apache/tika/eval/app/reports/ResultsReporter.java
+++ 
b/tika-eval/tika-eval-app/src/main/java/org/apache/tika/eval/app/reports/ResultsReporter.java
@@ -18,7 +18,6 @@ package org.apache.tika.eval.app.reports;
 
 
 import java.io.IOException;
-import java.io.InputStream;
 import java.nio.file.Files;
 import java.nio.file.Path;
 import java.nio.file.Paths;
@@ -32,7 +31,6 @@ import java.util.ArrayList;
 import java.util.HashMap;
 import java.util.List;
 import java.util.Map;
-import javax.xml.parsers.DocumentBuilder;
 
 import org.apache.commons.cli.CommandLine;
 import org.apache.commons.cli.DefaultParser;
@@ -86,11 +84,7 @@ public class ResultsReporter {
 
         ResultsReporter r = new ResultsReporter();
 
-        DocumentBuilder docBuilder = XMLReaderUtils.getDocumentBuilder();
-        Document doc;
-        try (InputStream is = Files.newInputStream(p)) {
-            doc = docBuilder.parse(is);
-        }
+        Document doc = XMLReaderUtils.buildDOM(p);
         Node docElement = doc.getDocumentElement();
         assert (docElement
                 .getNodeName()
diff --git 
a/tika-eval/tika-eval-structure/src/main/java/org/apache/tika/eval/structure/BlockExtractor.java
 
b/tika-eval/tika-eval-structure/src/main/java/org/apache/tika/eval/structure/BlockExtractor.java
index ef3965d8ae..c82f016b47 100644
--- 
a/tika-eval/tika-eval-structure/src/main/java/org/apache/tika/eval/structure/BlockExtractor.java
+++ 
b/tika-eval/tika-eval-structure/src/main/java/org/apache/tika/eval/structure/BlockExtractor.java
@@ -26,11 +26,11 @@ import java.util.Locale;
 import java.util.Set;
 
 import org.xml.sax.Attributes;
-import org.xml.sax.InputSource;
 import org.xml.sax.SAXException;
 import org.xml.sax.helpers.DefaultHandler;
 
 import org.apache.tika.exception.TikaException;
+import org.apache.tika.parser.ParseContext;
 import org.apache.tika.utils.XMLReaderUtils;
 
 /**
@@ -68,7 +68,7 @@ public final class BlockExtractor extends DefaultHandler {
     public static List<Block> extract(String xhtml) throws TikaException, 
SAXException,
             IOException {
         BlockExtractor extractor = new BlockExtractor();
-        XMLReaderUtils.getSAXParser().parse(new InputSource(new 
StringReader(xhtml)), extractor);
+        XMLReaderUtils.parseSAX(new StringReader(xhtml), extractor, new 
ParseContext());
         return extractor.blocks;
     }
 
diff --git a/tika-parent/forbidden-apis-xml-factory-bans.txt 
b/tika-parent/forbidden-apis-xml-factory-bans.txt
new file mode 100644
index 0000000000..65043421e1
--- /dev/null
+++ b/tika-parent/forbidden-apis-xml-factory-bans.txt
@@ -0,0 +1,81 @@
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements.  See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License.  You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+# TIKA-4939: every XML parser in main code comes from 
org.apache.tika.utils.XMLReaderUtils,
+# which is pooled, secured and covered by XmlSecurityContractTest. Building 
one anywhere
+# else bypasses all three. XMLReaderUtils itself is exempted with 
@SuppressForbidden.
+
+@defaultMessage build XML parsers through org.apache.tika.utils.XMLReaderUtils
+
+javax.xml.parsers.SAXParserFactory#newInstance()
+javax.xml.parsers.SAXParserFactory#newInstance(java.lang.String,java.lang.ClassLoader)
+javax.xml.parsers.SAXParserFactory#newDefaultInstance()
+javax.xml.parsers.SAXParserFactory#newDefaultNSInstance()
+javax.xml.parsers.SAXParserFactory#newNSInstance()
+javax.xml.parsers.SAXParserFactory#newNSInstance(java.lang.String,java.lang.ClassLoader)
+javax.xml.parsers.DocumentBuilderFactory#newInstance()
+javax.xml.parsers.DocumentBuilderFactory#newInstance(java.lang.String,java.lang.ClassLoader)
+javax.xml.parsers.DocumentBuilderFactory#newDefaultInstance()
+javax.xml.parsers.DocumentBuilderFactory#newDefaultNSInstance()
+javax.xml.parsers.DocumentBuilderFactory#newNSInstance()
+javax.xml.parsers.DocumentBuilderFactory#newNSInstance(java.lang.String,java.lang.ClassLoader)
+javax.xml.transform.TransformerFactory#newInstance()
+javax.xml.transform.TransformerFactory#newInstance(java.lang.String,java.lang.ClassLoader)
+javax.xml.transform.TransformerFactory#newDefaultInstance()
+javax.xml.stream.XMLInputFactory#newInstance()
+javax.xml.stream.XMLInputFactory#newFactory()
+javax.xml.stream.XMLInputFactory#newInstance(java.lang.String,java.lang.ClassLoader)
+javax.xml.stream.XMLInputFactory#newFactory(java.lang.String,java.lang.ClassLoader)
+javax.xml.stream.XMLInputFactory#newDefaultFactory()
+javax.xml.xpath.XPathFactory#newInstance()
+javax.xml.xpath.XPathFactory#newInstance(java.lang.String)
+javax.xml.xpath.XPathFactory#newInstance(java.lang.String,java.lang.String,java.lang.ClassLoader)
+javax.xml.xpath.XPathFactory#newDefaultInstance()
+javax.xml.validation.SchemaFactory#newInstance(java.lang.String)
+javax.xml.validation.SchemaFactory#newInstance(java.lang.String,java.lang.String,java.lang.ClassLoader)
+javax.xml.validation.SchemaFactory#newDefaultInstance()
+org.xml.sax.helpers.XMLReaderFactory#createXMLReader()
+org.xml.sax.helpers.XMLReaderFactory#createXMLReader(java.lang.String)
+org.apache.commons.xml.secure.SecureSAXParserFactory#newInstance()
+org.apache.commons.xml.secure.SecureSAXParserFactory#newInstance(java.lang.String,java.lang.ClassLoader)
+org.apache.commons.xml.secure.SecureSAXParserFactory#newNSInstance()
+org.apache.commons.xml.secure.SecureSAXParserFactory#newNSInstance(java.lang.String,java.lang.ClassLoader)
+org.apache.commons.xml.secure.SecureSAXParserFactory#newDefaultInstance()
+org.apache.commons.xml.secure.SecureSAXParserFactory#newDefaultNSInstance()
+org.apache.commons.xml.secure.SecureDocumentBuilderFactory#newInstance()
+org.apache.commons.xml.secure.SecureDocumentBuilderFactory#newInstance(java.lang.String,java.lang.ClassLoader)
+org.apache.commons.xml.secure.SecureDocumentBuilderFactory#newNSInstance()
+org.apache.commons.xml.secure.SecureDocumentBuilderFactory#newNSInstance(java.lang.String,java.lang.ClassLoader)
+org.apache.commons.xml.secure.SecureDocumentBuilderFactory#newDefaultInstance()
+org.apache.commons.xml.secure.SecureDocumentBuilderFactory#newDefaultNSInstance()
+org.apache.commons.xml.secure.SecureTransformerFactory#newInstance()
+org.apache.commons.xml.secure.SecureTransformerFactory#newInstance(java.lang.String,java.lang.ClassLoader)
+org.apache.commons.xml.secure.SecureTransformerFactory#newDefaultInstance()
+org.apache.commons.xml.secure.SecureXMLInputFactory#newFactory()
+org.apache.commons.xml.secure.SecureXMLInputFactory#newInstance()
+org.apache.commons.xml.secure.SecureXMLInputFactory#newFactory(java.lang.String,java.lang.ClassLoader)
+org.apache.commons.xml.secure.SecureXMLInputFactory#newDefaultFactory()
+
+# StAX is deprecated for removal in 5.0 (TIKA-4938); main code must not 
reintroduce it
+org.apache.tika.utils.XMLReaderUtils#getXMLInputFactory() @ StAX is 
deprecated; parse with XMLReaderUtils.parseSAX
+org.apache.tika.utils.XMLReaderUtils#getXMLInputFactory(org.apache.tika.parser.ParseContext)
 @ StAX is deprecated; parse with XMLReaderUtils.parseSAX
+
+# a raw parser honors a caller-supplied EntityResolver, and a resolver that 
answers with a
+# bare system id makes the parser fetch it (measured on the JDK and standalone 
Xerces);
+# parseSAX and buildDOM install Tika's resolver and never consult the caller's
+org.apache.tika.utils.XMLReaderUtils#getSAXParser() @ parse with 
XMLReaderUtils.parseSAX; a raw parser accepts a caller resolver that can fetch 
external entities
+org.apache.tika.utils.XMLReaderUtils#getXMLReader() @ parse with 
XMLReaderUtils.parseSAX; a raw reader accepts a caller resolver that can fetch 
external entities
+org.apache.tika.utils.XMLReaderUtils#getDocumentBuilder() @ parse with 
XMLReaderUtils.buildDOM, create with newDocument; a raw builder accepts a 
caller resolver that can fetch external entities
+org.apache.tika.utils.XMLReaderUtils#getDocumentBuilder(org.apache.tika.parser.ParseContext)
 @ parse with XMLReaderUtils.buildDOM; a raw builder accepts a caller resolver 
that can fetch external entities
diff --git a/tika-parent/pom.xml b/tika-parent/pom.xml
index 7b445da214..f665da3da4 100644
--- a/tika-parent/pom.xml
+++ b/tika-parent/pom.xml
@@ -325,6 +325,7 @@
     
<metadata.forbiddenapis.signaturesFile>${maven.multiModuleProjectDirectory}/tika-parent/forbidden-apis-metadata-string-key-bans.txt</metadata.forbiddenapis.signaturesFile>
     <!-- TIKA-4848: default-on in every module's main sources (see 
forbiddenapis-check below) -->
     
<exception.forbiddenapis.signaturesFile>${maven.multiModuleProjectDirectory}/tika-parent/forbidden-apis-exception-reporting-bans.txt</exception.forbiddenapis.signaturesFile>
+    
<xml.forbiddenapis.signaturesFile>${maven.multiModuleProjectDirectory}/tika-parent/forbidden-apis-xml-factory-bans.txt</xml.forbiddenapis.signaturesFile>
     <!-- Per-module extra main-source bans; a module overrides this (e.g. to
          ${metadata.forbiddenapis.signaturesFile}) instead of overriding the
          forbiddenapis-check execution, whose child list would REPLACE the 
parent's
@@ -1455,6 +1456,9 @@
         <version>${forbiddenapis.version}</version>
         <configuration>
           <targetVersion>${maven.compiler.target}</targetVersion>
+          <suppressAnnotations>
+            
<suppressAnnotation>org.apache.tika.utils.SuppressForbidden</suppressAnnotation>
+          </suppressAnnotations>
           
<ignoreSignaturesOfMissingClasses>true</ignoreSignaturesOfMissingClasses>
           <failOnUnsupportedJava>false</failOnUnsupportedJava>
           <excludes>test-documents/*.class</excludes>
@@ -1481,6 +1485,7 @@
             <configuration>
               <signaturesFiles>
                 
<signaturesFile>${exception.forbiddenapis.signaturesFile}</signaturesFile>
+                
<signaturesFile>${xml.forbiddenapis.signaturesFile}</signaturesFile>
                 
<signaturesFile>${forbiddenapis.module.signaturesFile}</signaturesFile>
               </signaturesFiles>
             </configuration>
diff --git 
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/geoinfo/GeographicInformationXxeTest.java
 
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/geoinfo/GeographicInformationXxeTest.java
new file mode 100644
index 0000000000..6270abc3b2
--- /dev/null
+++ 
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/geoinfo/GeographicInformationXxeTest.java
@@ -0,0 +1,142 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.parser.geoinfo;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.junit.jupiter.api.Assertions.fail;
+
+import java.io.FileNotFoundException;
+import java.io.IOException;
+import java.io.InputStream;
+import java.net.ConnectException;
+import java.net.InetAddress;
+import java.net.ServerSocket;
+import java.net.Socket;
+import java.nio.charset.StandardCharsets;
+import java.util.concurrent.atomic.AtomicInteger;
+
+import org.junit.jupiter.api.AfterAll;
+import org.junit.jupiter.api.BeforeAll;
+import org.junit.jupiter.api.Test;
+
+import org.apache.tika.TikaTest;
+import org.apache.tika.io.TikaInputStream;
+import org.apache.tika.metadata.Metadata;
+import org.apache.tika.parser.ParseContext;
+import org.apache.tika.sax.BodyContentHandler;
+
+/**
+ * Apache SIS parses ISO 19139 with its own StAX reader and resolves 
xlink:href. None of a
+ * DOCTYPE, an external entity, an xlink or a schemaLocation may reach the 
network or a file.
+ */
+public class GeographicInformationXxeTest extends TikaTest {
+
+    private static final String SECRET = "SECRET_CONTENT_19139";
+    private static ServerSocket oracle;
+    private static final AtomicInteger CONNECTIONS = new AtomicInteger();
+    private static String base;
+
+    @BeforeAll
+    public static void startOracle() throws IOException {
+        oracle = new ServerSocket(0, 50, InetAddress.getLoopbackAddress());
+        Thread t = new Thread(() -> {
+            while (!oracle.isClosed()) {
+                try (Socket s = oracle.accept()) {
+                    CONNECTIONS.incrementAndGet();
+                } catch (IOException e) {
+                    return;
+                }
+            }
+        }, "xxe-oracle");
+        t.setDaemon(true);
+        t.start();
+        base = "http://127.0.0.1:"; + oracle.getLocalPort() + "/";
+    }
+
+    @AfterAll
+    public static void stopOracle() throws IOException {
+        oracle.close();
+    }
+
+    private static String fixture() throws IOException {
+        try (InputStream is = GeographicInformationXxeTest.class
+                .getResourceAsStream("/test-documents/sampleFile.iso19139")) {
+            return new String(is.readAllBytes(), StandardCharsets.UTF_8);
+        }
+    }
+
+    private static String afterDeclaration(String xml, String insert) {
+        int end = xml.indexOf("?>") + 2;
+        return xml.substring(0, end) + insert + xml.substring(end);
+    }
+
+    private static String parse(String xml) throws Exception {
+        BodyContentHandler handler = new BodyContentHandler(-1);
+        try (TikaInputStream tis = 
TikaInputStream.get(xml.getBytes(StandardCharsets.UTF_8))) {
+            new GeographicInformationParser().parse(tis, handler, new 
Metadata(), new ParseContext());
+        }
+        return handler.toString();
+    }
+
+    private void assertNoFetch(String name, String xml) {
+        int before = CONNECTIONS.get();
+        String text = "";
+        try {
+            text = parse(xml);
+        } catch (Exception e) {
+            Throwable t = e;
+            while (t != null) {
+                if (t instanceof FileNotFoundException || t instanceof 
ConnectException ||
+                        (t.getMessage() != null && 
(t.getMessage().contains("couldnt_possibly_exist") ||
+                                t.getMessage().contains("Connection 
refused")))) {
+                    fail(name + ": parser tried to fetch an external 
resource", e);
+                }
+                t = t.getCause();
+            }
+        }
+        assertEquals(before, CONNECTIONS.get(), name + ": parser connected to 
the oracle");
+        assertFalse(text.contains(SECRET), name + ": leaked external content");
+    }
+
+    @Test
+    public void testExternalDtdAndEntity() throws Exception {
+        String xml = fixture();
+        assertNoFetch("dtd http", afterDeclaration(xml,
+                "<!DOCTYPE gmd:MD_Metadata SYSTEM \"" + base + "x.dtd\">"));
+        assertNoFetch("dtd file", afterDeclaration(xml,
+                "<!DOCTYPE gmd:MD_Metadata SYSTEM 
\"file:///couldnt_possibly_exist/x.dtd\">"));
+        assertNoFetch("entity http", afterDeclaration(xml,
+                "<!DOCTYPE gmd:MD_Metadata [<!ENTITY x SYSTEM \"" + base + 
"e.txt\">]>")
+                .replaceFirst("<gco:CharacterString>", 
"<gco:CharacterString>&x;"));
+        assertNoFetch("parameter entity", afterDeclaration(xml,
+                "<!DOCTYPE gmd:MD_Metadata [<!ENTITY % p SYSTEM \"" + base + 
"p.dtd\">%p;]>"));
+    }
+
+    @Test
+    public void testXlinkAndSchemaLocationNotResolved() throws Exception {
+        String xml = fixture();
+        assertTrue(xml.contains("<gmd:contact>"), "fixture shape changed");
+        assertNoFetch("xlink:href", xml.replaceFirst("<gmd:contact>",
+                "<gmd:contact xlink:href=\"" + base + "contact.xml\">"));
+        assertNoFetch("xlink:href empty element", 
xml.replaceFirst("<gmd:parentIdentifier>.*?</gmd:parentIdentifier>",
+                "<gmd:parentIdentifier xlink:href=\"" + base + 
"parent.xml\"/>"));
+        assertNoFetch("schemaLocation", 
xml.replaceFirst("xsi:schemaLocation=\"[^\"]*\"",
+                "xsi:schemaLocation=\"http://www.isotc211.org/2005/gmd " + 
base + "gmd.xsd\""));
+    }
+}
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/TestXMLEntityExpansion.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/TestXMLEntityExpansion.java
index a101390617..e8821e6bcf 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/TestXMLEntityExpansion.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/TestXMLEntityExpansion.java
@@ -38,28 +38,6 @@ import org.apache.tika.utils.XMLReaderUtils;
 
 public class TestXMLEntityExpansion extends XMLTestBase {
 
-    private static final byte[] ENTITY_EXPANSION_BOMB = new String(
-            "<!DOCTYPE kaboom [ " + "<!ENTITY a \"1234567890\" > " +
-                    "<!ENTITY b \"&a;&a;&a;&a;&a;&a;&a;&a;&a;&a;\" >" +
-                    "<!ENTITY c \"&b;&b;&b;&b;&b;&b;&b;&b;&b;&b;\" > " +
-                    "<!ENTITY d \"&c;&c;&c;&c;&c;&c;&c;&c;&c;&c;\" > " +
-                    "<!ENTITY e \"&d;&d;&d;&d;&d;&d;&d;&d;&d;&d;\" > " +
-                    "<!ENTITY f \"&e;&e;&e;&e;&e;&e;&e;&e;&e;&e;\" > " +
-                    "<!ENTITY g \"&f;&f;&f;&f;&f;&f;&f;&f;&f;&f;\" > " +
-                    "<!ENTITY h \"&g;&g;&g;&g;&g;&g;&g;&g;&g;&g;\" > " +
-                    "<!ENTITY i \"&h;&h;&h;&h;&h;&h;&h;&h;&h;&h;\" > " +
-                    "<!ENTITY j \"&i;&i;&i;&i;&i;&i;&i;&i;&i;&i;\" > " +
-                    "<!ENTITY k \"&j;&j;&j;&j;&j;&j;&j;&j;&j;&j;\" > " +
-                    "<!ENTITY l \"&k;&k;&k;&k;&k;&k;&k;&k;&k;&k;\" > " +
-                    "<!ENTITY m \"&l;&l;&l;&l;&l;&l;&l;&l;&l;&l;\" > " +
-                    "<!ENTITY n \"&m;&m;&m;&m;&m;&m;&m;&m;&m;&m;\" > " +
-                    "<!ENTITY o \"&n;&n;&n;&n;&n;&n;&n;&n;&n;&n;\" > " +
-                    "<!ENTITY p \"&o;&o;&o;&o;&o;&o;&o;&o;&o;&o;\" > " +
-                    "<!ENTITY q \"&p;&p;&p;&p;&p;&p;&p;&p;&p;&p;\" > " +
-                    "<!ENTITY r \"&q;&q;&q;&q;&q;&q;&q;&q;&q;&q;\" > " +
-                    "<!ENTITY s \"&r;&r;&r;&r;&r;&r;&r;&r;&r;&r;\" > " + "]> " 
+
-                    "<kaboom>&s;</kaboom>").getBytes(StandardCharsets.UTF_8);
-
     private static void test(String testFileName, byte[] bytes, Parser parser, 
ParseContext context)
             throws Exception {
         boolean ex = false;
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/TestXXEInXML.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/TestXXEInXML.java
new file mode 100644
index 0000000000..e2a52372cf
--- /dev/null
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/TestXXEInXML.java
@@ -0,0 +1,425 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.parser;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.junit.jupiter.api.Assertions.fail;
+
+import java.io.ByteArrayInputStream;
+import java.io.ByteArrayOutputStream;
+import java.io.FileNotFoundException;
+import java.io.IOException;
+import java.io.InputStream;
+import java.net.ConnectException;
+import java.net.ServerSocket;
+import java.net.Socket;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.concurrent.atomic.AtomicInteger;
+import javax.xml.parsers.SAXParserFactory;
+
+import org.apache.commons.io.IOUtils;
+import org.apache.pdfbox.pdmodel.PDDocument;
+import org.apache.pdfbox.pdmodel.PDPage;
+import org.apache.pdfbox.pdmodel.common.PDMetadata;
+import org.apache.pdfbox.pdmodel.common.PDStream;
+import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
+import org.apache.pdfbox.pdmodel.interactive.form.PDXFAResource;
+import org.junit.jupiter.api.AfterAll;
+import org.junit.jupiter.api.BeforeAll;
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.Timeout;
+import org.junit.jupiter.params.ParameterizedTest;
+import org.junit.jupiter.params.provider.ValueSource;
+import org.xml.sax.SAXException;
+import org.xml.sax.helpers.DefaultHandler;
+
+import org.apache.tika.io.TikaInputStream;
+import org.apache.tika.metadata.Metadata;
+import org.apache.tika.metadata.PDF;
+import org.apache.tika.metadata.TikaCoreProperties;
+import org.apache.tika.utils.XMLReaderUtils;
+
+/**
+ * Drives the document parsers, not XMLReaderUtils, with XXE and 
entity-expansion payloads
+ * injected into every XML-based format Tika reads: bare XML and its dialects, 
and the XML
+ * parts inside zip containers. The oracle for "fetched" is a local socket 
that must never
+ * see a connection, plus the file-not-found that a resolved bogus file URI 
would raise.
+ * XFA and XMP streams inside a synthesized PDF, and XMP packets inside image 
formats, are
+ * covered by injecting into the packet's own padding.
+ */
+public class TestXXEInXML extends XMLTestBase {
+
+    private static final String[] XML_FILES = {"testXXE.xml", 
"testWORD_2003ml.xml",
+            "testWORD_2006ml.xml", "testSVG.svg", "rsstest_20.rss", 
"testATOM.atom",
+            "testXLIFF12.xlf", "testTMX.tmx", "test.fb2", "testODTMacro.fodt"};
+
+    private static final String[] ZIP_FILES = {"testWORD.docx", 
"testWORD_macros.docm",
+            "testEXCEL_textbox.xlsx", "testEXCEL_macro.xlsm", 
"testPPT_2imgs.pptx",
+            "testPPT_macros.pptm", "testVISIO.vsdx", "testXPS_various.xps", 
"testEPUB.epub",
+            "testODTStyles2.odt", "testFooter.ods", "testMasterFooter.odp",
+            "testKeynote2018.key", "testPages2013.pages", 
"testNumbers2013.numbers"};
+
+    private static final String BOGUS_FILE = 
"file:///couldnt_possibly_exist/xxe.dtd";
+    private static final long MAX_CHARS = 50_000_000L;
+
+    private static ServerSocket oracle;
+    private static Thread acceptor;
+    private static final AtomicInteger CONNECTIONS = new AtomicInteger();
+    private static byte[] xxeHttp;
+    private static byte[] xxeFile;
+
+    @BeforeAll
+    public static void startOracle() throws IOException {
+        oracle = new ServerSocket(0, 50, 
java.net.InetAddress.getLoopbackAddress());
+        acceptor = new Thread(() -> {
+            while (!oracle.isClosed()) {
+                try (Socket s = oracle.accept()) {
+                    CONNECTIONS.incrementAndGet();
+                } catch (IOException e) {
+                    return;
+                }
+            }
+        }, "xxe-oracle");
+        acceptor.setDaemon(true);
+        acceptor.start();
+        String base = "http://127.0.0.1:"; + oracle.getLocalPort() + "/";
+        xxeHttp = ("<!DOCTYPE roottag SYSTEM \"" + base + "xxe.dtd\" [<!ENTITY 
% p SYSTEM \"" +
+                base + "p.dtd\">%p;]>").getBytes(StandardCharsets.UTF_8);
+        xxeFile = ("<!DOCTYPE roottag SYSTEM \"" + BOGUS_FILE + "\" [<!ENTITY 
% p SYSTEM \"" +
+                BOGUS_FILE + "\">%p;]>").getBytes(StandardCharsets.UTF_8);
+    }
+
+    @AfterAll
+    public static void stopOracle() throws IOException {
+        oracle.close();
+    }
+
+    // the oracle must catch a parser that does fetch, or every green test 
below is vacuous
+    @Test
+    public void testOracleDetectsFetch() throws Exception {
+        byte[] doc = injectXML(bareXml(), xxeHttp);
+        SAXParserFactory factory = SAXParserFactory.newInstance();
+        
factory.setFeature("http://apache.org/xml/features/nonvalidating/load-external-dtd";,
 true);
+        int before = CONNECTIONS.get();
+        try {
+            factory.newSAXParser().parse(new ByteArrayInputStream(doc), new 
DefaultHandler());
+        } catch (SAXException | IOException e) {
+            // the oracle closes the connection without answering; an error is 
expected
+        }
+        assertTrue(CONNECTIONS.get() > before, "unsecured parser did not reach 
the oracle");
+    }
+
+    @ParameterizedTest
+    @ValueSource(strings = {"http", "file"})
+    public void testXmlFiles(String payload) throws Exception {
+        for (String fileName : XML_FILES) {
+            byte[] injected = injectXML(read(fileName), payload(payload));
+            for (int i = 0; i < XMLReaderUtils.getPoolSize() + 1; i++) {
+                assertNoFetch(fileName, () -> parseBytes(fileName, injected));
+            }
+        }
+    }
+
+    // every XML plist carries an Apple DTD reference; dd-plist parses it with 
its own parser
+    @Test
+    public void testPlistDoctype() throws Exception {
+        String plist = "<?xml version=\"1.0\" encoding=\"UTF-8\"?><!DOCTYPE 
plist SYSTEM \"" +
+                new String(xxeHttp, 
StandardCharsets.UTF_8).replaceAll(".*SYSTEM \"([^\"]+)\".*", "$1") +
+                "\"><plist 
version=\"1.0\"><dict><key>k</key><string>v</string></dict></plist>";
+        byte[] bytes = plist.getBytes(StandardCharsets.UTF_8);
+        assertNoFetch("inline.plist", () -> parseBytes("inline.plist", bytes));
+    }
+
+    // an XInclude is a fetch vector only if a parser enables it; none may
+    @Test
+    public void testXIncludeNotResolved() throws Exception {
+        String base = new String(xxeHttp, 
StandardCharsets.UTF_8).replaceAll(".*SYSTEM \"([^\"]+)/xxe.dtd\".*", "$1");
+        byte[] doc = ("<?xml version=\"1.0\"?><r 
xmlns:xi=\"http://www.w3.org/2001/XInclude\";>" +
+                "<xi:include href=\"" + base + "/inc.xml\"/><xi:include 
href=\"" + BOGUS_FILE + "\" parse=\"text\"/></r>")
+                .getBytes(StandardCharsets.UTF_8);
+        assertNoFetch("xinclude.xml", () -> parseBytes("xinclude.xml", doc));
+    }
+
+    @ParameterizedTest
+    @ValueSource(strings = {"http", "file"})
+    public void testZipContainers(String payload) throws Exception {
+        for (String fileName : ZIP_FILES) {
+            Path injected = injectZip(fileName, payload(payload));
+            try {
+                assertNoFetch(fileName, () -> parsePath(fileName, injected));
+            } finally {
+                Files.delete(injected);
+            }
+        }
+    }
+
+    @Test
+    @Timeout(300)
+    public void testExpansionBombInXmlFiles() throws Exception {
+        for (String fileName : XML_FILES) {
+            byte[] injected = injectXML(read(fileName), ENTITY_EXPANSION_BOMB);
+            assertBounded(fileName, () -> parseBytes(fileName, injected));
+        }
+    }
+
+    @Test
+    @Timeout(600)
+    public void testExpansionBombInZipContainers() throws Exception {
+        for (String fileName : ZIP_FILES) {
+            Path injected = injectZip(fileName, ENTITY_EXPANSION_BOMB);
+            try {
+                assertBounded(fileName, () -> parsePath(fileName, injected));
+            } finally {
+                Files.delete(injected);
+            }
+        }
+    }
+
+    // CVE-2025-66516: the XFA stream inside a PDF is XML parsed straight out 
of the file
+    @ParameterizedTest
+    @ValueSource(strings = {"http", "file"})
+    public void testXfaInPdf(String payload) throws Exception {
+        byte[] pdf = pdfWithXfa(injectXML(xfaXml(), payload(payload)));
+        for (int i = 0; i < XMLReaderUtils.getPoolSize() + 1; i++) {
+            Metadata metadata = new Metadata();
+            assertNoFetch("xfa.pdf", () -> parseBytes("xfa.pdf", pdf, 
metadata));
+            assertEquals("true", metadata.get(PDF.HAS_XFA), "the XFA stream 
was not reached");
+        }
+    }
+
+    @Test
+    @Timeout(120)
+    public void testExpansionBombInXfa() throws Exception {
+        byte[] pdf = pdfWithXfa(injectXML(xfaXml(), ENTITY_EXPANSION_BOMB));
+        Metadata metadata = new Metadata();
+        assertBounded("xfa.pdf", () -> parseBytes("xfa.pdf", pdf, metadata));
+        assertEquals("true", metadata.get(PDF.HAS_XFA), "the XFA stream was 
not reached");
+    }
+
+    private static byte[] pdfWithXfa(byte[] xfa) throws IOException {
+        try (PDDocument doc = new PDDocument()) {
+            doc.addPage(new PDPage());
+            PDAcroForm form = new PDAcroForm(doc);
+            doc.getDocumentCatalog().setAcroForm(form);
+            PDStream stream = new PDStream(doc, new ByteArrayInputStream(xfa));
+            form.setXFA(new PDXFAResource(stream.getCOSObject()));
+            ByteArrayOutputStream bos = new ByteArrayOutputStream();
+            doc.save(bos);
+            return bos.toByteArray();
+        }
+    }
+
+    private static byte[] xfaXml() {
+        return ("<?xml version=\"1.0\" encoding=\"UTF-8\"?>" +
+                "<xdp:xdp xmlns:xdp=\"http://ns.adobe.com/xdp/\";>" +
+                "<template 
xmlns=\"http://www.xfa.org/schema/xfa-template/3.3/\";>" +
+                "<subform name=\"form1\"><field 
name=\"n\"><assist><toolTip>Name</toolTip>" +
+                
"</assist></field></subform></template></xdp:xdp>").getBytes(StandardCharsets.UTF_8);
+    }
+
+    // XMP packets inside binary containers: the payload goes in after the 
xpacket PI and
+    // the same number of padding bytes comes out before the closing PI, so no 
offset moves
+    private static final String[] XMP_FILES = {"testJPEG_GEO.jpg", 
"testTIFF.tif", "testPSD_xmp.psd", "testJXL_ISOBMFF.jxl"};
+
+    @ParameterizedTest
+    @ValueSource(strings = {"http", "file"})
+    public void testXmpInBinaryFormats(String payload) throws Exception {
+        for (String fileName : XMP_FILES) {
+            byte[] original = read(fileName);
+            Metadata clean = new Metadata();
+            parseBytes(fileName, original, clean);
+            assertTrue(hasXmpKey(clean), fileName + ": the clean fixture 
yields no XMP metadata");
+            byte[] injected = injectIntoXmpPacket(original, payload(payload));
+            assertNoFetch(fileName, () -> parseBytes(fileName, injected));
+        }
+    }
+
+    @Test
+    @Timeout(300)
+    public void testExpansionBombInXmp() throws Exception {
+        for (String fileName : XMP_FILES) {
+            byte[] injected = injectIntoXmpPacket(read(fileName), 
ENTITY_EXPANSION_BOMB);
+            assertBounded(fileName, () -> parseBytes(fileName, injected));
+        }
+    }
+
+    @ParameterizedTest
+    @ValueSource(strings = {"http", "file"})
+    public void testXmpInPdf(String payload) throws Exception {
+        byte[] pdf = pdfWithXmp(injectXML(xmpPacket(), payload(payload)));
+        Metadata metadata = new Metadata();
+        assertNoFetch("xmp.pdf", () -> parseBytes("xmp.pdf", pdf, metadata));
+        assertEquals("true", metadata.get(PDF.HAS_XMP), "the XMP stream was 
not reached");
+    }
+
+    private static boolean hasXmpKey(Metadata metadata) {
+        for (String name : metadata.names()) {
+            if (name.startsWith("xmp") || name.startsWith("dc:")) {
+                return true;
+            }
+        }
+        return false;
+    }
+
+    static byte[] injectIntoXmpPacket(byte[] file, byte[] payload) {
+        String s = new String(file, StandardCharsets.ISO_8859_1);
+        int begin = s.indexOf("<?xpacket begin");
+        assertTrue(begin >= 0, "no XMP packet");
+        int insert = s.indexOf("?>", begin) + 2;
+        int end = s.indexOf("<?xpacket end", insert);
+        int pad = 0;
+        while (pad < end - insert && Character.isWhitespace(s.charAt(end - 1 - 
pad))) {
+            pad++;
+        }
+        assertTrue(pad >= payload.length, "packet padding " + pad + " < 
payload " + payload.length);
+        byte[] out = new byte[file.length];
+        System.arraycopy(file, 0, out, 0, insert);
+        System.arraycopy(payload, 0, out, insert, payload.length);
+        System.arraycopy(file, insert, out, insert + payload.length, end - 
payload.length - insert);
+        System.arraycopy(file, end, out, end, file.length - end);
+        return out;
+    }
+
+    private static byte[] xmpPacket() {
+        return ("<?xpacket begin=\"\" id=\"W5M0MpCehiHzreSzNTczkc9d\"?>" +
+                "<x:xmpmeta xmlns:x=\"adobe:ns:meta/\"><rdf:RDF 
xmlns:rdf=\"http://www.w3.org/1999/02/22-rdf-syntax-ns#\";>" +
+                "<rdf:Description rdf:about=\"\" 
xmlns:dc=\"http://purl.org/dc/elements/1.1/\";>" +
+                "<dc:title><rdf:Alt><rdf:li 
xml:lang=\"x-default\">t</rdf:li></rdf:Alt></dc:title>" +
+                "</rdf:Description></rdf:RDF></x:xmpmeta><?xpacket 
end=\"w\"?>").getBytes(StandardCharsets.UTF_8);
+    }
+
+    private static byte[] pdfWithXmp(byte[] xmp) throws IOException {
+        try (PDDocument doc = new PDDocument()) {
+            doc.addPage(new PDPage());
+            doc.getDocumentCatalog().setMetadata(new PDMetadata(doc, new 
ByteArrayInputStream(xmp)));
+            ByteArrayOutputStream bos = new ByteArrayOutputStream();
+            doc.save(bos);
+            return bos.toByteArray();
+        }
+    }
+
+    private interface Parse {
+        long run() throws Exception;
+    }
+
+    private static void assertNoFetch(String fileName, Parse parse) {
+        int before = CONNECTIONS.get();
+        try {
+            parse.run();
+        } catch (Exception e) {
+            // injection may well corrupt the document; only a fetch is a 
failure
+            assertNotAFetch(fileName, e);
+        }
+        assertEquals(before, CONNECTIONS.get(), fileName + ": parser connected 
to the oracle");
+    }
+
+    private static void assertBounded(String fileName, Parse parse) {
+        try {
+            long chars = parse.run();
+            assertTrue(chars < MAX_CHARS, fileName + ": expanded " + chars + " 
chars");
+        } catch (Exception e) {
+            Throwable t = e;
+            while (t != null) {
+                if (t.getMessage() != null && 
t.getMessage().startsWith("harness:")) {
+                    fail(fileName + ": " + t.getMessage());
+                }
+                t = t.getCause();
+            }
+        }
+    }
+
+    private static void assertNotAFetch(String fileName, Exception e) {
+        Throwable t = e;
+        while (t != null) {
+            if (t instanceof FileNotFoundException || t instanceof 
ConnectException) {
+                fail(fileName + ": parser tried to fetch an external 
resource", e);
+            }
+            String msg = t.getMessage();
+            if (msg != null && (msg.contains("couldnt_possibly_exist") ||
+                    msg.contains("No such file") || msg.contains("Connection 
refused"))) {
+                fail(fileName + ": parser tried to fetch an external 
resource", e);
+            }
+            t = t.getCause();
+        }
+    }
+
+    private static byte[] payload(String name) {
+        return "http".equals(name) ? xxeHttp : xxeFile;
+    }
+
+    private static byte[] bareXml() {
+        return "<?xml version=\"1.0\" 
encoding=\"UTF-8\"?><document>blah</document>"
+                .getBytes(StandardCharsets.UTF_8);
+    }
+
+    private static byte[] read(String fileName) throws IOException {
+        try (InputStream is = 
TestXXEInXML.class.getResourceAsStream("/test-documents/" + fileName)) {
+            assertTrue(is != null, "missing fixture " + fileName);
+            ByteArrayOutputStream bos = new ByteArrayOutputStream();
+            IOUtils.copy(is, bos);
+            return bos.toByteArray();
+        }
+    }
+
+    private static Path injectZip(String fileName, byte[] payload) throws 
IOException {
+        try (TikaInputStream tis = TikaInputStream.get(
+                TestXXEInXML.class.getResourceAsStream("/test-documents/" + 
fileName))) {
+            return injectZippedXMLs(tis.getPath(), payload);
+        }
+    }
+
+    private static long parseBytes(String fileName, byte[] bytes) throws 
Exception {
+        return parseBytes(fileName, bytes, new Metadata());
+    }
+
+    private static long parseBytes(String fileName, byte[] bytes, Metadata 
metadata)
+            throws Exception {
+        try (TikaInputStream tis = TikaInputStream.get(bytes)) {
+            return parseCounting(fileName, tis, metadata);
+        }
+    }
+
+    private static long parsePath(String fileName, Path path) throws Exception 
{
+        try (TikaInputStream tis = TikaInputStream.get(path)) {
+            return parseCounting(fileName, tis, new Metadata());
+        }
+    }
+
+    private static long parseCounting(String fileName, TikaInputStream tis, 
Metadata metadata)
+            throws Exception {
+        CountingHandler handler = new CountingHandler();
+        metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
+        AUTO_DETECT_PARSER.parse(tis, handler, metadata, new ParseContext());
+        return handler.count;
+    }
+
+    // stops a runaway expansion instead of waiting for the timeout
+    private static class CountingHandler extends DefaultHandler {
+        long count = 0;
+
+        @Override
+        public void characters(char[] ch, int start, int length) throws 
SAXException {
+            count += length;
+            if (count > MAX_CHARS) {
+                throw new SAXException("harness: more than " + MAX_CHARS + " 
characters expanded");
+            }
+        }
+    }
+}
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/XMLTestBase.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/XMLTestBase.java
index 13468a25eb..1c73d49a47 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/XMLTestBase.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/XMLTestBase.java
@@ -17,9 +17,8 @@
 package org.apache.tika.parser;
 
 import java.io.ByteArrayOutputStream;
-import java.io.File;
-import java.io.FileOutputStream;
 import java.io.IOException;
+import java.nio.charset.StandardCharsets;
 import java.nio.file.Files;
 import java.nio.file.Path;
 import java.util.Collections;
@@ -48,6 +47,28 @@ import org.apache.tika.sax.TextContentHandler;
 
 public class XMLTestBase extends TikaTest {
 
+    static final byte[] ENTITY_EXPANSION_BOMB = new String(
+            "<!DOCTYPE kaboom [ " + "<!ENTITY a \"1234567890\" > " +
+                    "<!ENTITY b \"&a;&a;&a;&a;&a;&a;&a;&a;&a;&a;\" >" +
+                    "<!ENTITY c \"&b;&b;&b;&b;&b;&b;&b;&b;&b;&b;\" > " +
+                    "<!ENTITY d \"&c;&c;&c;&c;&c;&c;&c;&c;&c;&c;\" > " +
+                    "<!ENTITY e \"&d;&d;&d;&d;&d;&d;&d;&d;&d;&d;\" > " +
+                    "<!ENTITY f \"&e;&e;&e;&e;&e;&e;&e;&e;&e;&e;\" > " +
+                    "<!ENTITY g \"&f;&f;&f;&f;&f;&f;&f;&f;&f;&f;\" > " +
+                    "<!ENTITY h \"&g;&g;&g;&g;&g;&g;&g;&g;&g;&g;\" > " +
+                    "<!ENTITY i \"&h;&h;&h;&h;&h;&h;&h;&h;&h;&h;\" > " +
+                    "<!ENTITY j \"&i;&i;&i;&i;&i;&i;&i;&i;&i;&i;\" > " +
+                    "<!ENTITY k \"&j;&j;&j;&j;&j;&j;&j;&j;&j;&j;\" > " +
+                    "<!ENTITY l \"&k;&k;&k;&k;&k;&k;&k;&k;&k;&k;\" > " +
+                    "<!ENTITY m \"&l;&l;&l;&l;&l;&l;&l;&l;&l;&l;\" > " +
+                    "<!ENTITY n \"&m;&m;&m;&m;&m;&m;&m;&m;&m;&m;\" > " +
+                    "<!ENTITY o \"&n;&n;&n;&n;&n;&n;&n;&n;&n;&n;\" > " +
+                    "<!ENTITY p \"&o;&o;&o;&o;&o;&o;&o;&o;&o;&o;\" > " +
+                    "<!ENTITY q \"&p;&p;&p;&p;&p;&p;&p;&p;&p;&p;\" > " +
+                    "<!ENTITY r \"&q;&q;&q;&q;&q;&q;&q;&q;&q;&q;\" > " +
+                    "<!ENTITY s \"&r;&r;&r;&r;&r;&r;&r;&r;&r;&r;\" > " + "]> " 
+
+                    "<kaboom>&s;</kaboom>").getBytes(StandardCharsets.UTF_8);
+
     static byte[] injectXML(byte[] input, byte[] toInject) throws IOException {
 
         int startXML = -1;
@@ -70,33 +91,30 @@ public class XMLTestBase extends TikaTest {
         return bos.toByteArray();
     }
 
-    static Path injectZippedXMLs(Path original, byte[] toInject, boolean 
includeSlides)
-            throws IOException {
-        ZipFile input = new ZipFile(original.toFile());
-        File output = Files.createTempFile("tika-xxe-", ".zip").toFile();
-        ZipOutputStream outZip = new ZipOutputStream(new 
FileOutputStream(output));
-        Enumeration<? extends ZipEntry> zipEntryEnumeration = input.entries();
-        while (zipEntryEnumeration.hasMoreElements()) {
-            ZipEntry entry = zipEntryEnumeration.nextElement();
-            ByteArrayOutputStream bos = new ByteArrayOutputStream();
-            IOUtils.copy(input.getInputStream(entry), bos);
-            byte[] bytes = bos.toByteArray();
-            if (entry.getName().endsWith(".xml") &&
-                    //don't inject the slides because you'll get a bean 
exception
-                    //Unexpected node
-                    (!includeSlides && 
!entry.getName().contains("slides/slide"))) {
-                bytes = injectXML(bytes, toInject);
+    // XML parts of zip containers: OOXML/ODF/EPUB parts, OOXML relationships, 
XPS pages
+    static boolean isXmlEntry(String name) {
+        return name.endsWith(".xml") || name.endsWith(".rels") || 
name.endsWith(".fpage");
+    }
+
+    static Path injectZippedXMLs(Path original, byte[] toInject) throws 
IOException {
+        Path output = Files.createTempFile("tika-xxe-", ".zip");
+        try (ZipFile input = new ZipFile(original.toFile());
+                ZipOutputStream outZip = new 
ZipOutputStream(Files.newOutputStream(output))) {
+            Enumeration<? extends ZipEntry> zipEntryEnumeration = 
input.entries();
+            while (zipEntryEnumeration.hasMoreElements()) {
+                ZipEntry entry = zipEntryEnumeration.nextElement();
+                ByteArrayOutputStream bos = new ByteArrayOutputStream();
+                IOUtils.copy(input.getInputStream(entry), bos);
+                byte[] bytes = bos.toByteArray();
+                if (isXmlEntry(entry.getName())) {
+                    bytes = injectXML(bytes, toInject);
+                }
+                outZip.putNextEntry(new ZipEntry(entry.getName()));
+                outZip.write(bytes);
+                outZip.closeEntry();
             }
-            ZipEntry outEntry = new ZipEntry(entry.getName());
-            outZip.putNextEntry(outEntry);
-            outZip.write(bytes);
-            outZip.closeEntry();
         }
-        input.close();
-        outZip.flush();
-        outZip.close();
-
-        return output.toPath();
+        return output;
     }
 
     static void parse(String testFileName, TikaInputStream is, Parser parser, 
ParseContext context)
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/main/java/org/apache/tika/parser/microsoft/ooxml/xwpf/ml2006/Word2006MLParser.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/main/java/org/apache/tika/parser/microsoft/ooxml/xwpf/ml2006/Word2006MLParser.java
index de16d84057..43fbf1b79c 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/main/java/org/apache/tika/parser/microsoft/ooxml/xwpf/ml2006/Word2006MLParser.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/main/java/org/apache/tika/parser/microsoft/ooxml/xwpf/ml2006/Word2006MLParser.java
@@ -56,12 +56,9 @@ public class Word2006MLParser extends AbstractOfficeParser {
         xhtml.startDocument();
         tis.setCloseShield();
         try {
-            //need to get new SAXParser because
-            //an attachment might require another SAXParser
-            //mid-parse
-            XMLReaderUtils.getSAXParser().parse(tis,
-                    new EmbeddedContentHandler(
-                            new Word2006MLDocHandler(xhtml, metadata, 
context)));
+            XMLReaderUtils.parseSAX(tis,
+                    new EmbeddedContentHandler(new Word2006MLDocHandler(xhtml, 
metadata, context)),
+                    context);
         } catch (SAXException e) {
             throw new TikaException("XML parse error", e);
         } finally {
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/main/java/org/apache/tika/parser/microsoft/xml/AbstractXML2003Parser.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/main/java/org/apache/tika/parser/microsoft/xml/AbstractXML2003Parser.java
index 0810cfb4ec..a98801255b 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/main/java/org/apache/tika/parser/microsoft/xml/AbstractXML2003Parser.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/main/java/org/apache/tika/parser/microsoft/xml/AbstractXML2003Parser.java
@@ -100,12 +100,9 @@ public abstract class AbstractXML2003Parser implements 
Parser {
         TaggedContentHandler tagged = new TaggedContentHandler(balancer);
         tis.setCloseShield();
         try {
-            //need to get new SAXParser because
-            //an attachment might require another SAXParser
-            //mid-parse
-            XMLReaderUtils.getSAXParser().parse(tis,
-                    new EmbeddedContentHandler(
-                            getContentHandler(tagged, metadata, context)));
+            XMLReaderUtils.parseSAX(tis,
+                    new EmbeddedContentHandler(getContentHandler(tagged, 
metadata, context)),
+                    context);
         } catch (SAXException e) {
             WriteLimitReachedException.throwIfWriteLimitReached(e);
             // Close anything the aborted parse left open, then propagate.
diff --git 
a/tika-pipes/tika-async-cli/src/main/java/org/apache/tika/async/cli/TikaConfigAsyncWriter.java
 
b/tika-pipes/tika-async-cli/src/main/java/org/apache/tika/async/cli/TikaConfigAsyncWriter.java
index 020a5dce98..45eac4c380 100644
--- 
a/tika-pipes/tika-async-cli/src/main/java/org/apache/tika/async/cli/TikaConfigAsyncWriter.java
+++ 
b/tika-pipes/tika-async-cli/src/main/java/org/apache/tika/async/cli/TikaConfigAsyncWriter.java
@@ -71,7 +71,7 @@ class TikaConfigAsyncWriter {
                         simpleAsyncConfig.getTikaConfig());
             }
         } else {
-            document = XMLReaderUtils.getDocumentBuilder().newDocument();
+            document = XMLReaderUtils.newDocument();
             properties = document.createElement("properties");
             document.appendChild(properties);
         }

Reply via email to