This is an automated email from the ASF dual-hosted git repository.
tballison pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/tika.git
The following commit(s) were added to refs/heads/main by this push:
new ae72b5cca2 TIKA-4920: fix ci to actually catch locale issues (#3229)
ae72b5cca2 is described below
commit ae72b5cca25bb92e75314bc8c5b964d82cb0b7a8
Author: Tim Allison <[email protected]>
AuthorDate: Thu Sep 24 06:30:34 2026 -0400
TIKA-4920: fix ci to actually catch locale issues (#3229)
---
.github/workflows/main-jdk17-locale-build.yml | 12 ++++---
.github/workflows/main-jdk17-windows-build.yml | 17 +++++-----
.../org/apache/tika/DefaultLocaleCanaryTest.java | 37 ++++++++++++++++++++++
.../ml/junkdetect/tools/BoundaryBigramAudit.java | 5 +--
.../ml/junkdetect/tools/BuildJunkTrainingData.java | 3 +-
.../ml/junkdetect/tools/LineScriptFractions.java | 5 +--
.../tika/ml/junkdetect/tools/TrainJunkModel.java | 5 +--
.../tools/BuildJunkAugmentationData.java | 9 +++---
.../tika/parser/ocr/TesseractOCRParserTest.java | 3 ++
.../org/apache/tika/parser/ogg/FlacParserTest.java | 3 ++
.../tika/parser/microsoft/JackcessParserTest.java | 3 ++
.../tika/pipes/ignite/IgniteConfigStoreTest.java | 3 ++
12 files changed, 82 insertions(+), 23 deletions(-)
diff --git a/.github/workflows/main-jdk17-locale-build.yml
b/.github/workflows/main-jdk17-locale-build.yml
index a336032fe4..d46b5de978 100644
--- a/.github/workflows/main-jdk17-locale-build.yml
+++ b/.github/workflows/main-jdk17-locale-build.yml
@@ -21,9 +21,11 @@
# Linux, not Windows: locale bugs are JVM-level, so the 2x-cost Windows runner
# buys nothing here -- main-jdk17-windows-build covers the OS-specific surface.
#
-# Locale via -Duser.language/-Duser.country, not LANG/LC_ALL: the JVM silently
-# falls back to en_US when the named locale is not generated on the runner, so
-# the env-var form can pass while testing nothing.
+# Locale via JAVA_TOOL_OPTIONS, which every JVM (surefire forks included)
reads at
+# startup. Not LANG/LC_ALL: the JVM falls back to en_US when the locale is not
+# generated on the runner. Not -Duser.language on the mvn command line:
surefire
+# sets that in the fork only after the default Locale is fixed, so it tests
nothing.
+# DefaultLocaleCanaryTest fails the build if the locale does not reach the
fork.
#
# push-only, like the jdk21/jdk25 builds: locale regressions are rare and not
# usually PR-specific, so a full reactor build per PR is not worth the cost.
@@ -66,9 +68,11 @@ jobs:
# build runs them. If a new testcontainers module appears, add it to
this list --
# forgetting only makes this job slower, it does not weaken it.
- name: Build with Maven (tr_TR locale)
+ env:
+ JAVA_TOOL_OPTIONS: -Duser.language=tr -Duser.country=TR
+ TIKA_EXPECTED_LOCALE: tr-TR
run: |
mvn install -Pfast -pl :tika-annotation-processor -am -B -q
mvn clean test install -Pci -T1C \
-pl
'!:tika-pipes-es-integration-tests,!:tika-pipes-kafka-integration-tests,!:tika-pipes-opensearch-integration-tests,!:tika-pipes-s3-integration-tests,!:tika-pipes-solr-integration-tests'
\
- -Duser.language=tr -Duser.country=TR \
-B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git a/.github/workflows/main-jdk17-windows-build.yml
b/.github/workflows/main-jdk17-windows-build.yml
index ebd0f251a1..bea2ad2dc2 100644
--- a/.github/workflows/main-jdk17-windows-build.yml
+++ b/.github/workflows/main-jdk17-windows-build.yml
@@ -19,10 +19,11 @@
# - path handling: the checkout dir below deliberately contains a space
# - an alternate (non-en_US) locale
#
-# Locale is set with -Duser.language/-Duser.country, NOT LANG/LC_ALL. Those env
-# vars are POSIX-only: the Windows JVM reads the OS locale via Win32 and
ignores
-# them, and even on Linux the JVM silently falls back to en_US when the named
-# locale is not generated on the box. -D always applies.
+# Locale is set with JAVA_TOOL_OPTIONS, which every JVM (surefire forks
included)
+# reads at startup. NOT LANG/LC_ALL: the Windows JVM reads the OS locale via
Win32
+# and ignores them. NOT -Duser.language on the mvn command line: surefire sets
that
+# in the fork only after the default Locale is fixed, so the tests ran in
en_US.
+# DefaultLocaleCanaryTest fails the build if the locale does not reach the
fork.
#
# push-only, like the jdk21/jdk25 and tr_TR builds. This is the most expensive
# job in CI -- a full reactor on a 2x-cost runner, ~48 min -- and it gated PR
@@ -66,11 +67,11 @@ jobs:
restore-keys: maven-${{ runner.os }}-
- name: Build with Maven (de_DE locale)
working-directory: 'tika build dir'
- # The -Duser.* args MUST stay quoted: PowerShell splits an unquoted
dotted
- # -D property before it reaches mvn, so the locale silently never
applies
- # and the leftover fragment fails the build as an unknown lifecycle
phase.
+ env:
+ JAVA_TOOL_OPTIONS: -Duser.language=de -Duser.country=DE
+ TIKA_EXPECTED_LOCALE: de-DE
# No -T1C here (TIKA-4867): this is the heaviest job (locale + e2e +
javadoc)
# on a 4-core runner; parallel modules starved the tika-server
integration
# tests into startup timeouts. Serial reactor also needs no annotation-
# processor pre-install.
- run: mvn clean test install javadoc:aggregate -Pci -Pe2e
"-Duser.language=de" "-Duser.country=DE" -B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
+ run: mvn clean test install javadoc:aggregate -Pci -Pe2e -B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git
a/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java
b/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java
new file mode 100644
index 0000000000..e02b5a2c1f
--- /dev/null
+++ b/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java
@@ -0,0 +1,37 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+
+import java.util.Locale;
+
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.condition.EnabledIfEnvironmentVariable;
+
+/**
+ * Fails a locale CI job whose locale never reached the forked test JVM:
-Duser.language on
+ * the mvn command line becomes a system property only after the default
Locale is fixed.
+ */
+public class DefaultLocaleCanaryTest {
+
+ @Test
+ @EnabledIfEnvironmentVariable(named = "TIKA_EXPECTED_LOCALE", matches =
".+")
+ public void testDefaultLocaleMatchesExpected() {
+ assertEquals(System.getenv("TIKA_EXPECTED_LOCALE"),
Locale.getDefault().toLanguageTag());
+ }
+}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
index f64986b8dd..4936e70886 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
@@ -24,6 +24,7 @@ import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
import java.util.HashMap;
+import java.util.Locale;
import java.util.stream.Stream;
import java.util.zip.GZIPInputStream;
@@ -72,7 +73,7 @@ public final class BoundaryBigramAudit {
for (Path file : files) {
String fname = file.getFileName().toString();
String name = fname.substring(0, fname.length() -
".train.gz".length())
- .toUpperCase();
+ .toUpperCase(Locale.ROOT);
Character.UnicodeScript target;
try {
target = Character.UnicodeScript.valueOf(name);
@@ -135,7 +136,7 @@ public final class BoundaryBigramAudit {
int distAsciiDrop = distinctKeptUnderAsciiDrop.size();
System.out.printf("%-22s %,14d %,14d %,14d %,14d %,12d | %,14d
%,14d%n",
- name.toLowerCase(), inS, boundary, foreign, asciiRun,
total,
+ name.toLowerCase(Locale.ROOT), inS, boundary, foreign,
asciiRun, total,
distAll - distForeignDrop, distAll - distAsciiDrop);
}
}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
index b4460501b4..8806be34aa 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
@@ -31,6 +31,7 @@ import java.util.Collections;
import java.util.HashMap;
import java.util.LinkedHashMap;
import java.util.List;
+import java.util.Locale;
import java.util.Map;
import java.util.Random;
import java.util.TreeMap;
@@ -409,7 +410,7 @@ public class BuildJunkTrainingData {
List<String> dev = sentences.subList(nTrain, nTrain + nDev);
List<String> test = sentences.subList(nTrain + nDev,
sentences.size());
- String baseName = script.toLowerCase();
+ String baseName = script.toLowerCase(Locale.ROOT);
writeGzipped(outputDir.resolve(baseName + ".train.gz"), train);
writeGzipped(outputDir.resolve(baseName + ".dev.gz"), dev);
writeGzipped(outputDir.resolve(baseName + ".test.gz"), test);
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
index bcda57c9f7..21cdefa954 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
@@ -23,6 +23,7 @@ import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.Paths;
+import java.util.Locale;
import java.util.zip.GZIPInputStream;
/**
@@ -77,7 +78,7 @@ public final class LineScriptFractions {
for (Path file : files) {
String fname = file.getFileName().toString();
String name = fname.substring(0, fname.length() -
".train.gz".length())
- .toUpperCase();
+ .toUpperCase(Locale.ROOT);
Character.UnicodeScript target = mapScript(name);
if (target == null) {
System.out.printf("%-20s (no UnicodeScript mapping for
'%s')%n", name, name);
@@ -135,7 +136,7 @@ public final class LineScriptFractions {
long below5 = bucketCounts[0];
System.out.printf("%-20s %,10d %,10d |%s%n",
- name.toLowerCase(), lines, below5, sb.toString());
+ name.toLowerCase(Locale.ROOT), lines, below5,
sb.toString());
}
}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
index 13cbc20381..4bebad6fa3 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
@@ -30,6 +30,7 @@ import java.util.Collections;
import java.util.HashMap;
import java.util.LinkedHashMap;
import java.util.List;
+import java.util.Locale;
import java.util.Map;
import java.util.Random;
import java.util.TreeMap;
@@ -289,7 +290,7 @@ public class TrainJunkModel {
allTrainFiles.add(trainFile);
String filename = trainFile.getFileName().toString();
String script = filename.substring(0, filename.length() -
".train.gz".length())
- .toUpperCase();
+ .toUpperCase(Locale.ROOT);
trainFilePaths.put(script, trainFile);
tallyFileBuckets(trainFile, pairsByScript, unigramsByScript,
totalsByScript);
}
@@ -947,7 +948,7 @@ public class TrainJunkModel {
String filename = trainFile.getFileName().toString();
String script = filename.endsWith(".train.gz")
- ? filename.substring(0, filename.length() -
".train.gz".length()).toUpperCase()
+ ? filename.substring(0, filename.length() -
".train.gz".length()).toUpperCase(Locale.ROOT)
: null;
if (script == null || !pairsByScript.containsKey(script)) {
// Fall back to the bucket with the most bigrams.
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
b/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
index b5b76b198e..c135f9357e 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
@@ -31,6 +31,7 @@ import java.util.Collections;
import java.util.HashMap;
import java.util.LinkedHashMap;
import java.util.List;
+import java.util.Locale;
import java.util.Map;
import java.util.Random;
import java.util.TreeMap;
@@ -336,7 +337,7 @@ public final class BuildJunkAugmentationData {
continue;
}
String scriptName = ds.script.name();
- if (!baselineLineCounts.containsKey(scriptName.toLowerCase()))
{
+ if
(!baselineLineCounts.containsKey(scriptName.toLowerCase(Locale.ROOT))) {
// No baseline bucket for this script — nothing to augment.
droppedNoBaseline++;
continue;
@@ -392,7 +393,7 @@ public final class BuildJunkAugmentationData {
String script = entry.getKey();
List<String> chunks = entry.getValue();
int docs = scriptDocCount.getOrDefault(script, 0);
- long baselineLines =
baselineLineCounts.getOrDefault(script.toLowerCase(), 0L);
+ long baselineLines =
baselineLineCounts.getOrDefault(script.toLowerCase(Locale.ROOT), 0L);
long fracCapVal = (long) Math.floor(baselineLines * fracCap);
long cap = Math.min(hardCap, fracCapVal);
@@ -463,7 +464,7 @@ public final class BuildJunkAugmentationData {
Path dst = outputDir.resolve(name);
if (name.endsWith(".train.gz")) {
String script = name.substring(0, name.length() -
".train.gz".length())
- .toUpperCase();
+ .toUpperCase(Locale.ROOT);
List<String> add = finalLines.get(script);
if (add != null && !add.isEmpty()) {
rewriteTrainWithAppend(src, dst, add);
@@ -740,7 +741,7 @@ public final class BuildJunkAugmentationData {
int idxLang = -1;
int idxLangId = -1;
for (int i = 0; i < cols.length; i++) {
- switch (cols[i].toUpperCase()) {
+ switch (cols[i].toUpperCase(Locale.ROOT)) {
case "FILE_PATH":
idxPath = i;
break;
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
index 5430b082ec..4685dacb14 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
@@ -31,6 +31,7 @@ import java.util.Map;
import com.fasterxml.jackson.databind.ObjectMapper;
import org.junit.jupiter.api.Disabled;
import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.condition.DisabledIfSystemProperty;
import org.apache.tika.TikaTest;
import org.apache.tika.config.ParseContextConfig;
@@ -221,6 +222,8 @@ public class TesseractOCRParserTest extends TikaTest {
}
+ // TODO TIKA-4923: metadata-extractor lowercases the resolution unit in
the default locale
+ @DisabledIfSystemProperty(named = "user.language", matches = "tr")
@Test
public void getNormalMetadataToo() throws Exception {
//this should be successful whether or not TesseractOCR is
installed/active
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-audiovideo-module/src/test/java/org/apache/tika/parser/ogg/FlacParserTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-audiovideo-module/src/test/java/org/apache/tika/parser/ogg/FlacParserTest.java
index 4fde9e14ee..f4398d2307 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-audiovideo-module/src/test/java/org/apache/tika/parser/ogg/FlacParserTest.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-audiovideo-module/src/test/java/org/apache/tika/parser/ogg/FlacParserTest.java
@@ -26,6 +26,7 @@ import java.nio.file.Path;
import java.util.List;
import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.condition.DisabledIfSystemProperty;
import org.junit.jupiter.api.io.TempDir;
import org.apache.tika.TikaTest;
@@ -88,6 +89,8 @@ public class FlacParserTest extends TikaTest {
* both sources are merged before the pick (both are front covers here,
* so the first one, from the comment, wins).
*/
+ // TODO TIKA-4921: vorbis-java lowercases comment keys in the default
locale
+ @DisabledIfSystemProperty(named = "user.language", matches = "tr")
@Test
public void testCommentAndNativePictureYieldOneThumbnail() throws
Exception {
List<Metadata> metadataList =
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
index abd20d7def..f2d9240fb6 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
@@ -23,6 +23,7 @@ import static org.junit.jupiter.api.Assertions.assertTrue;
import java.util.List;
import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.condition.DisabledIfSystemProperty;
import org.apache.tika.TikaTest;
import org.apache.tika.exception.EncryptedDocumentException;
@@ -79,6 +80,8 @@ public class JackcessParserTest extends TikaTest {
}
}
+ // TODO TIKA-4924: jackcess-encrypt uppercases cipher params in the
default locale
+ @DisabledIfSystemProperty(named = "user.language", matches = "tr")
@Test
public void testPassword() throws Exception {
ParseContext c = new ParseContext();
diff --git
a/tika-pipes/tika-pipes-config-store-ignite/src/test/java/org/apache/tika/pipes/ignite/IgniteConfigStoreTest.java
b/tika-pipes/tika-pipes-config-store-ignite/src/test/java/org/apache/tika/pipes/ignite/IgniteConfigStoreTest.java
index f55a9a5145..bb31368ab3 100644
---
a/tika-pipes/tika-pipes-config-store-ignite/src/test/java/org/apache/tika/pipes/ignite/IgniteConfigStoreTest.java
+++
b/tika-pipes/tika-pipes-config-store-ignite/src/test/java/org/apache/tika/pipes/ignite/IgniteConfigStoreTest.java
@@ -33,6 +33,7 @@ import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.BeforeEach;
import org.junit.jupiter.api.Disabled;
import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.condition.DisabledIfSystemProperty;
import org.junit.jupiter.api.io.TempDir;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
@@ -43,6 +44,8 @@ import org.apache.tika.plugins.ExtensionConfig;
/**
* Integration tests for {@link IgniteConfigStore} using an embedded Ignite
3.x server.
*/
+// TODO TIKA-4922: Ignite uppercases unquoted table names in the default locale
+@DisabledIfSystemProperty(named = "user.language", matches = "tr")
public class IgniteConfigStoreTest {
private static final Logger LOG =
LoggerFactory.getLogger(IgniteConfigStoreTest.class);