This is an automated email from the ASF dual-hosted git repository.

tballison pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/tika.git


The following commit(s) were added to refs/heads/main by this push:
     new 3db3c2e87f TIKA-4920: run tests under a random default locale; forks 
inherit the… (#3247)
3db3c2e87f is described below

commit 3db3c2e87f8bd19362145611b5043872663938a7
Author: Tim Allison <[email protected]>
AuthorDate: Tue Oct 6 17:34:52 2026 -0400

    TIKA-4920: run tests under a random default locale; forks inherit the… 
(#3247)
---
 .github/workflows/main-jdk17-build.yml             |   3 +-
 .github/workflows/main-jdk17-locale-build.yml      |  78 -------------
 .github/workflows/main-jdk17-windows-build.yml     |  20 ++--
 .github/workflows/main-jdk21-build.yml             |   2 +-
 .github/workflows/main-jdk25-build.yml             |   2 +-
 .skills/devs/development/SKILL.md                  |   6 +
 CHANGES.txt                                        |   8 ++
 pom.xml                                            |   1 +
 .../java/org/apache/tika/utils/ProcessUtils.java   |  36 ++++++
 .../org/apache/tika/DefaultLocaleCanaryTest.java   |  37 -------
 .../apache/tika/test/RandomLocaleListenerTest.java |  64 +++++++++++
 tika-ml/tika-ml-chardetect/pom.xml                 |   7 --
 .../chardetect/tools/BuildCharsetTrainingData.java |  35 +++---
 .../chardetect/tools/DiagnoseDiscrimination.java   |   2 +-
 tika-ml/tika-ml-junkdetect-tools/pom.xml           |   7 --
 .../ml/junkdetect/tools/BoundaryBigramAudit.java   |   4 +-
 .../ml/junkdetect/tools/BuildJunkTrainingData.java |  40 +++----
 .../tika/ml/junkdetect/tools/DebugScriptRuns.java  |   7 +-
 .../ml/junkdetect/tools/LineScriptFractions.java   |   8 +-
 .../tika/ml/junkdetect/tools/ScriptCensus.java     |   9 +-
 .../tika/ml/junkdetect/tools/TrainJunkModel.java   |  34 +++---
 .../tools/BuildJunkAugmentationData.java           |  34 +++---
 tika-ml/tika-ml-junkdetect/pom.xml                 |   7 --
 .../ml/junkdetect/LatinSiblingComparisonTest.java  |   9 +-
 tika-parent/checkstyle.xml                         |   4 +-
 tika-parent/pom.xml                                |  27 ++++-
 .../apache/tika/parser/grib/GribParserTest.java    |   8 ++
 .../tika/parser/netcdf/NetCDFParserTest.java       |   8 ++
 .../tika/parser/ocr/TesseractOCRParserTest.java    |   4 +-
 .../org/apache/tika/parser/sas/SAS7BDATParser.java |   4 +-
 .../apache/tika/parser/sas/SAS7BDATParserTest.java |   8 +-
 .../tika/parser/image/ImageMetadataExtractor.java  |  44 ++++++--
 .../image/ImageMetadataExtractorLocaleTest.java    |  36 ++++++
 .../parser/image/ImageMetadataExtractorTest.java   |  23 +---
 .../tika/parser/microsoft/JackcessParserTest.java  |   4 +-
 .../tika/pipes/core/PerClientServerManager.java    |   9 ++
 .../tika/pipes/core/SharedServerManager.java       |   9 ++
 .../core/PerClientServerManagerLocaleTest.java     |  70 ++++++++++++
 tika-test-support/pom.xml                          |  56 ++++++++++
 .../org/apache/tika/test/RandomLocaleListener.java | 122 +++++++++++++++++++++
 ...junit.platform.launcher.LauncherSessionListener |  17 +++
 ...g.junit.platform.launcher.TestExecutionListener |  17 +++
 42 files changed, 651 insertions(+), 279 deletions(-)

diff --git a/.github/workflows/main-jdk17-build.yml 
b/.github/workflows/main-jdk17-build.yml
index ffb7f16445..033a5b6888 100644
--- a/.github/workflows/main-jdk17-build.yml
+++ b/.github/workflows/main-jdk17-build.yml
@@ -100,6 +100,7 @@ jobs:
         run: |
           mvn install -Pfast -pl :tika-annotation-processor -am -B -q
           mvn clean test install ${{ github.event_name == 'push' && 
'javadoc:aggregate' || '' }} -Pci -T1C \
+            -Dtika.test.locale=random \
             -pl "$(echo "$IT_MODULES" | sed 's/:/!:/g')" \
             -B 
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
 
@@ -140,7 +141,7 @@ jobs:
       # so without this they would drop out of license checking entirely.
       - name: Run integration tests
         run: |
-          mvn clean apache-rat:check test -Pci \
+          mvn clean apache-rat:check test -Pci -Dtika.test.locale=random \
             -pl "$IT_MODULES" \
             -B 
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
 
diff --git a/.github/workflows/main-jdk17-locale-build.yml 
b/.github/workflows/main-jdk17-locale-build.yml
deleted file mode 100644
index 899bcd5687..0000000000
--- a/.github/workflows/main-jdk17-locale-build.yml
+++ /dev/null
@@ -1,78 +0,0 @@
-#
-# Licensed to the Apache Software Foundation (ASF) under one or more
-# contributor license agreements.  See the NOTICE file distributed with
-# this work for additional information regarding copyright ownership.
-# The ASF licenses this file to You under the Apache License, Version 2.0
-# (the "License"); you may not use this file except in compliance with
-# the License.  You may obtain a copy of the License at
-#
-#      http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-#
-
-# tr_TR is the highest-value locale to test in Java: dotless-i means
-# "TIFF".toLowerCase() is not "tiff" unless the call passes Locale.ROOT.
-#
-# Linux, not Windows: locale bugs are JVM-level, so the 2x-cost Windows runner
-# buys nothing here -- main-jdk17-windows-build covers the OS-specific surface.
-#
-# Locale via JAVA_TOOL_OPTIONS, which every JVM (surefire forks included) 
reads at
-# startup. Not LANG/LC_ALL: the JVM falls back to en_US when the locale is not
-# generated on the runner. Not -Duser.language on the mvn command line: 
surefire
-# sets that in the fork only after the default Locale is fixed, so it tests 
nothing.
-# DefaultLocaleCanaryTest fails the build if the locale does not reach the 
fork.
-#
-# push-only, like the jdk21/jdk25 builds: locale regressions are rare and not
-# usually PR-specific, so a full reactor build per PR is not worth the cost.
-name: main jdk17 locale build (tr_TR)
-
-on:
-  push:
-    branches: [ main ]
-    paths-ignore:
-      - 'docs/**'
-
-jobs:
-  build:
-    runs-on: ubuntu-latest
-    timeout-minutes: 60
-    strategy:
-      matrix:
-        java: [ '17' ]
-
-    steps:
-      - uses: actions/checkout@v7
-      - name: Set up JDK ${{ matrix.java }}
-        uses: actions/setup-java@v6
-        with:
-          distribution: 'temurin'
-          java-version: ${{ matrix.java }}
-      - name: Cache Maven repository
-        # Explicit, not setup-java's cache: 'maven', so our own snapshots stay 
out of the tarball.
-        uses: actions/cache@v6
-        with:
-          path: |
-            ~/.m2/repository
-            !~/.m2/repository/org/apache/tika
-          key: maven-${{ runner.os }}-${{ hashFiles('**/pom.xml') }}
-          restore-keys: maven-${{ runner.os }}-
-      - name: Install external tools
-        run: sudo apt-get update && sudo apt-get install -y ffmpeg 
libimage-exiftool-perl
-      # The Docker-backed integration tests spin up 
Elasticsearch/OpenSearch/Solr/Kafka/RustFS
-      # for ~7.5 min and carry no locale signal, so they are excluded here; 
the main jdk17
-      # build runs them. If a new testcontainers module appears, add it to 
this list --
-      # forgetting only makes this job slower, it does not weaken it.
-      - name: Build with Maven (tr_TR locale)
-        env:
-          JAVA_TOOL_OPTIONS: -Duser.language=tr -Duser.country=TR
-          TIKA_EXPECTED_LOCALE: tr-TR
-        run: |
-          mvn install -Pfast -pl :tika-annotation-processor -am -B -q
-          mvn clean test install -Pci -T1C \
-            -pl 
'!:tika-pipes-es-integration-tests,!:tika-pipes-kafka-integration-tests,!:tika-pipes-opensearch-integration-tests,!:tika-pipes-s3-integration-tests,!:tika-pipes-solr-integration-tests'
 \
-            -B 
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git a/.github/workflows/main-jdk17-windows-build.yml 
b/.github/workflows/main-jdk17-windows-build.yml
index 4613496528..ac8de6c2b5 100644
--- a/.github/workflows/main-jdk17-windows-build.yml
+++ b/.github/workflows/main-jdk17-windows-build.yml
@@ -17,20 +17,21 @@
 
 # The one Windows job. Two things it uniquely covers:
 #   - path handling: the checkout dir below deliberately contains a space
-#   - an alternate (non-en_US) locale
+#   - child JVMs (tika-server, pipes forks) started under a non-en_US locale
 #
-# Locale is set with JAVA_TOOL_OPTIONS, which every JVM (surefire forks 
included)
-# reads at startup. NOT LANG/LC_ALL: the Windows JVM reads the OS locale via 
Win32
-# and ignores them. NOT -Duser.language on the mvn command line: surefire sets 
that
-# in the fork only after the default Locale is fixed, so the tests ran in 
en_US.
-# DefaultLocaleCanaryTest fails the build if the locale does not reach the 
fork.
+# The Linux jobs run in-process tests under a random default locale
+# (-Dtika.test.locale=random, RandomLocaleListener); this one runs them under
+# de_DE. Pipes forks receive their parent's locale as -D flags, but JVMs the
+# tests start themselves (tika-server) inherit only the environment, hence
+# JAVA_TOOL_OPTIONS here, which every JVM reads at startup.
+# NOT LANG/LC_ALL: the Windows JVM reads the OS locale via Win32 and ignores 
them.
 #
-# push-only, like the jdk21/jdk25 and tr_TR builds. This is the most expensive
+# push-only, like the jdk21/jdk25 builds. This is the most expensive
 # job in CI -- a full reactor on a 2x-cost runner, ~48 min -- and it gated PR
 # wall clock all by itself while every Linux job finished in ~22. Windows-only
 # regressions are real but rare, so they are caught on main within the hour
 # rather than paid for on every PR push.
-name: main jdk17 windows build (de_DE)
+name: main jdk17 windows build
 
 on:
   push:
@@ -65,11 +66,10 @@ jobs:
             !~/.m2/repository/org/apache/tika
           key: maven-${{ runner.os }}-${{ hashFiles('**/pom.xml') }}
           restore-keys: maven-${{ runner.os }}-
-      - name: Build with Maven (de_DE locale)
+      - name: Build with Maven (de_DE child JVMs)
         working-directory: 'tika build dir'
         env:
           JAVA_TOOL_OPTIONS: -Duser.language=de -Duser.country=DE
-          TIKA_EXPECTED_LOCALE: de-DE
         # No -T1C here (TIKA-4867): this is the heaviest job (locale + e2e + 
javadoc)
         # on a 4-core runner; parallel modules starved the tika-server 
integration
         # tests into startup timeouts. Serial reactor also needs no annotation-
diff --git a/.github/workflows/main-jdk21-build.yml 
b/.github/workflows/main-jdk21-build.yml
index dba342210d..3aa717983a 100644
--- a/.github/workflows/main-jdk21-build.yml
+++ b/.github/workflows/main-jdk21-build.yml
@@ -48,4 +48,4 @@ jobs:
           key: maven-${{ runner.os }}-${{ hashFiles('**/pom.xml') }}
           restore-keys: maven-${{ runner.os }}-
       - name: Build with Maven
-        run: mvn install -Pfast -pl :tika-annotation-processor -am -B -q && 
mvn clean test install javadoc:aggregate -Pci -T1C -B 
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
+        run: mvn install -Pfast -pl :tika-annotation-processor -am -B -q && 
mvn clean test install javadoc:aggregate -Pci -T1C -Dtika.test.locale=random -B 
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git a/.github/workflows/main-jdk25-build.yml 
b/.github/workflows/main-jdk25-build.yml
index 690304a24d..58187e295b 100644
--- a/.github/workflows/main-jdk25-build.yml
+++ b/.github/workflows/main-jdk25-build.yml
@@ -48,4 +48,4 @@ jobs:
           key: maven-${{ runner.os }}-${{ hashFiles('**/pom.xml') }}
           restore-keys: maven-${{ runner.os }}-
       - name: Build with Maven
-        run: mvn install -Pfast -pl :tika-annotation-processor -am -B -q && 
mvn clean test install javadoc:aggregate -Pci -T1C -B 
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
+        run: mvn install -Pfast -pl :tika-annotation-processor -am -B -q && 
mvn clean test install javadoc:aggregate -Pci -T1C -Dtika.test.locale=random -B 
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git a/.skills/devs/development/SKILL.md 
b/.skills/devs/development/SKILL.md
index 9a04a78968..863c05aeb9 100644
--- a/.skills/devs/development/SKILL.md
+++ b/.skills/devs/development/SKILL.md
@@ -211,6 +211,12 @@ them back. Anything hot enough to need a real seek gets a 
channel from
   CONCATENATE-only bugs).
 - Keep tests non-duplicative: don't add a test whose failure another test
   already guarantees.
+- CI runs every test JVM under a random default locale 
(`-Dtika.test.locale=random`;
+  `RandomLocaleListener` in `tika-test-support`, wired through tika-parent). A 
plain
+  build keeps the JVM's own locale. A CI failure's stack trace and the fork's 
stderr
+  name the locale; reproduce with `-Dtika.test.locale=<tag>`.  A test that 
fails only
+  in some locales is a bug in the code under test (use `Locale.ROOT`), not a
+  reason to pin `Locale.US` in the test.
 - Where there's bang for the buck, prefer parameterized tests over
   copy-pasted cases, randomized inputs over hand-picked ones (log the seed
   so failures reproduce), and fuzzing for parsers and format/boundary
diff --git a/CHANGES.txt b/CHANGES.txt
index ef89983799..5a2ece3ec1 100644
--- a/CHANGES.txt
+++ b/CHANGES.txt
@@ -240,6 +240,14 @@ Release 4.1.0 - 9/26/2026
    * The Ignite config store quotes its table name, so lookups work under any
      default locale (TIKA-4922).
 
+   * CI runs each test JVM under a random default locale 
(-Dtika.test.locale=random,
+     tika-test-support's RandomLocaleListener); a failure names the locale and
+     -Dtika.test.locale=<tag> reproduces it. Replaces the fixed tr_TR CI job 
(TIKA-4920).
+
+   * Pipes forks now start with the parent JVM's default locale (-Duser.*)
+     unless forkedJvmArgs set user.language; a fresh JVM took the OS locale,
+     so forked parsing could differ from in-process parsing (TIKA-4920).
+
    * tika-annotation-processor is no longer a transitive dependency of
      tika-pipes-core or the tika-langdetect modules (provided scope).
 
diff --git a/pom.xml b/pom.xml
index 52b72c875b..8d9db9bb8b 100644
--- a/pom.xml
+++ b/pom.xml
@@ -37,6 +37,7 @@
   <modules>
     <module>tika-parent</module>
     <module>tika-bom</module>
+    <module>tika-test-support</module>
     <module>tika-core</module>
     <module>tika-annotation-processor</module>
     <module>tika-serialization</module>
diff --git a/tika-core/src/main/java/org/apache/tika/utils/ProcessUtils.java 
b/tika-core/src/main/java/org/apache/tika/utils/ProcessUtils.java
index e1916b43cc..244742cb74 100644
--- a/tika-core/src/main/java/org/apache/tika/utils/ProcessUtils.java
+++ b/tika-core/src/main/java/org/apache/tika/utils/ProcessUtils.java
@@ -20,6 +20,9 @@ package org.apache.tika.utils;
 import java.io.IOException;
 import java.nio.file.Files;
 import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
 import java.util.Optional;
 import java.util.UUID;
 import java.util.concurrent.ConcurrentHashMap;
@@ -84,6 +87,39 @@ public class ProcessUtils {
         return arg;
     }
 
+    /**
+     * The -D arguments that give a child JVM {@code locale} as its default: a 
fresh JVM reads
+     * the OS locale, not its parent's, and a command-line -D beats 
JAVA_TOOL_OPTIONS.
+     * Empty for {@link Locale#ROOT}.
+     */
+    public static List<String> defaultLocaleJvmArgs(Locale locale) {
+        List<String> args = new ArrayList<>();
+        if (locale.getLanguage().isEmpty()) {
+            return args;
+        }
+        args.add("-Duser.language=" + locale.getLanguage());
+        if (!locale.getScript().isEmpty()) {
+            args.add("-Duser.script=" + locale.getScript());
+        }
+        if (!locale.getCountry().isEmpty()) {
+            args.add("-Duser.country=" + locale.getCountry());
+        }
+        if (!locale.getVariant().isEmpty()) {
+            args.add("-Duser.variant=" + locale.getVariant());
+        }
+        StringBuilder extensions = new StringBuilder();
+        for (char key : locale.getExtensionKeys()) {
+            if (extensions.length() > 0) {
+                extensions.append('-');
+            }
+            
extensions.append(key).append('-').append(locale.getExtension(key));
+        }
+        if (extensions.length() > 0) {
+            args.add("-Duser.extensions=" + extensions);
+        }
+        return args;
+    }
+
     public static String unescapeCommandLine(String arg) {
         if (arg.contains(" ") && SystemUtils.IS_OS_WINDOWS &&
                 (arg.startsWith("\"") && arg.endsWith("\""))) {
diff --git 
a/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java 
b/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java
deleted file mode 100644
index e02b5a2c1f..0000000000
--- a/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java
+++ /dev/null
@@ -1,37 +0,0 @@
-/*
- * Licensed to the Apache Software Foundation (ASF) under one or more
- * contributor license agreements.  See the NOTICE file distributed with
- * this work for additional information regarding copyright ownership.
- * The ASF licenses this file to You under the Apache License, Version 2.0
- * (the "License"); you may not use this file except in compliance with
- * the License.  You may obtain a copy of the License at
- *
- *     http://www.apache.org/licenses/LICENSE-2.0
- *
- * Unless required by applicable law or agreed to in writing, software
- * distributed under the License is distributed on an "AS IS" BASIS,
- * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
- * See the License for the specific language governing permissions and
- * limitations under the License.
- */
-package org.apache.tika;
-
-import static org.junit.jupiter.api.Assertions.assertEquals;
-
-import java.util.Locale;
-
-import org.junit.jupiter.api.Test;
-import org.junit.jupiter.api.condition.EnabledIfEnvironmentVariable;
-
-/**
- * Fails a locale CI job whose locale never reached the forked test JVM: 
-Duser.language on
- * the mvn command line becomes a system property only after the default 
Locale is fixed.
- */
-public class DefaultLocaleCanaryTest {
-
-    @Test
-    @EnabledIfEnvironmentVariable(named = "TIKA_EXPECTED_LOCALE", matches = 
".+")
-    public void testDefaultLocaleMatchesExpected() {
-        assertEquals(System.getenv("TIKA_EXPECTED_LOCALE"), 
Locale.getDefault().toLanguageTag());
-    }
-}
diff --git 
a/tika-core/src/test/java/org/apache/tika/test/RandomLocaleListenerTest.java 
b/tika-core/src/test/java/org/apache/tika/test/RandomLocaleListenerTest.java
new file mode 100644
index 0000000000..eeec5d1982
--- /dev/null
+++ b/tika-core/src/test/java/org/apache/tika/test/RandomLocaleListenerTest.java
@@ -0,0 +1,64 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.test;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+import static org.junit.jupiter.api.Assertions.assertSame;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+import java.util.List;
+import java.util.Locale;
+
+import org.junit.jupiter.api.Test;
+
+/**
+ * Also the canary: the listener reaches every module through tika-parent, and 
this test fails
+ * the build if it stops doing so.
+ */
+public class RandomLocaleListenerTest {
+
+    @Test
+    public void testListenerAppliedToThisJvm() {
+        String tag = System.getProperty(RandomLocaleListener.PROPERTY);
+        assertNotNull(tag, "listener did not run");
+        assertEquals(tag, Locale.getDefault().toLanguageTag());
+    }
+
+    @Test
+    public void testChoose() {
+        assertEquals(Locale.forLanguageTag("tr-TR"), 
RandomLocaleListener.choose("tr-TR"));
+        assertSame(Locale.getDefault(), RandomLocaleListener.choose("system"));
+        assertSame(Locale.getDefault(), RandomLocaleListener.choose(null));
+        assertSame(Locale.getDefault(), RandomLocaleListener.choose(""));
+        assertThrows(IllegalArgumentException.class, () -> 
RandomLocaleListener.choose("?"));
+        List<Locale> candidates = RandomLocaleListener.candidates();
+        assertTrue(candidates.size() > 500, "only " + candidates.size() + " 
candidates");
+        
assertTrue(candidates.containsAll(RandomLocaleListener.PRIORITY_LOCALES));
+        // the tag printed on failure must reproduce the pick, variants and 
extensions included
+        for (Locale candidate : candidates) {
+            assertFalse(candidate.getLanguage().isEmpty());
+            assertEquals(candidate, 
RandomLocaleListener.choose(candidate.toLanguageTag()),
+                    candidate.toString());
+        }
+        for (int i = 0; i < 20; i++) {
+            
assertTrue(candidates.contains(RandomLocaleListener.choose("random")));
+        }
+    }
+}
diff --git a/tika-ml/tika-ml-chardetect/pom.xml 
b/tika-ml/tika-ml-chardetect/pom.xml
index 8ee7abe212..15f943f4a9 100644
--- a/tika-ml/tika-ml-chardetect/pom.xml
+++ b/tika-ml/tika-ml-chardetect/pom.xml
@@ -96,13 +96,6 @@
         </configuration>
       </plugin>
       <!-- Tools use System.out/printf freely — suppress forbidden-apis for 
this module -->
-      <plugin>
-        <groupId>de.thetaphi</groupId>
-        <artifactId>forbiddenapis</artifactId>
-        <configuration>
-          <skip>true</skip>
-        </configuration>
-      </plugin>
     </plugins>
   </build>
 
diff --git 
a/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/BuildCharsetTrainingData.java
 
b/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/BuildCharsetTrainingData.java
index f863456a69..fecb75fe5c 100644
--- 
a/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/BuildCharsetTrainingData.java
+++ 
b/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/BuildCharsetTrainingData.java
@@ -38,6 +38,7 @@ import java.util.HashMap;
 import java.util.HashSet;
 import java.util.LinkedHashMap;
 import java.util.List;
+import java.util.Locale;
 import java.util.Map;
 import java.util.Random;
 import java.util.Set;
@@ -513,19 +514,19 @@ public class BuildCharsetTrainingData {
         System.out.println("=== BuildCharsetTrainingData ===");
         System.out.println("  madlad-dir:          " + madladDir);
         System.out.println("  output-dir:          " + outputDir);
-        System.out.printf ("  sample cap:          %,d%n", sampleCap);
-        System.out.printf ("  byte budget:         %,d%n", byteBudget);
-        System.out.printf ("  chunk:               %d–%d bytes  seed=%d%n", 
minChunk, maxChunk, seed);
-        System.out.printf ("  unicode langs:       %d (from %s)%n",
+        System.out.printf(Locale.ROOT, "  sample cap:          %,d%n", 
sampleCap);
+        System.out.printf(Locale.ROOT, "  byte budget:         %,d%n", 
byteBudget);
+        System.out.printf(Locale.ROOT, "  chunk:               %d–%d bytes  
seed=%d%n", minChunk, maxChunk, seed);
+        System.out.printf(Locale.ROOT, "  unicode langs:       %d (from %s)%n",
                 unicodeLangs.size(), unicodeLangsFile.getFileName());
-        System.out.printf ("  charsets:            %d%n%n", 
targetCharsets.size());
+        System.out.printf(Locale.ROOT, "  charsets:            %d%n%n", 
targetCharsets.size());
 
         // Ambiguity gate: for each SBCS charset, precompute encoders for all
         // other SBCS charsets. A chunk is dropped if any rival produces
         // byte-for-byte identical output — such chunks carry no discriminative
         // signal and actively confuse the model.
         Map<String, List<CharsetEncoder>> sbcsRivals = 
buildSbcsRivals(targetCharsets);
-        System.out.printf("  ambiguity-gate:      %d SBCS charsets compared 
pairwise%n%n",
+        System.out.printf(Locale.ROOT, "  ambiguity-gate:      %d SBCS 
charsets compared pairwise%n%n",
                 sbcsRivals.size());
 
         // charset label → split name → sample count (for manifest)
@@ -540,7 +541,7 @@ public class BuildCharsetTrainingData {
             String[] splits       = {"train", "devtest", "test"};
             int[]    splitCaps    = {sampleCap, sampleCap, sampleCap};
             long[]   splitBudgets = {byteBudget, byteBudget / 5, byteBudget / 
5};
-            System.out.printf("%s  (%s)%s%n", label, javaName,
+            System.out.printf(Locale.ROOT, "%s  (%s)%s%n", label, javaName,
                     structOnly ? "  [structural-only: skipping train]" : "");
 
             // Determine contributing languages.
@@ -568,17 +569,17 @@ public class BuildCharsetTrainingData {
                     ? Math.max(5_000, UNICODE_SENTENCE_BUDGET / nLangs)
                     : Math.min(MAX_LOAD_CAP_PER_LANG, LEGACY_SENTENCE_BUDGET / 
nLangs);
             List<String> allSentences = new ArrayList<>();
-            System.out.printf("  Contributing languages (%d), 
perLangCap=%,d:%n",
+            System.out.printf(Locale.ROOT, "  Contributing languages (%d), 
perLangCap=%,d:%n",
                     nLangs, perLangCap);
             for (String lang : langs) {
                 Path langDir = madladDir.resolve(lang);
                 List<String> sents = loadMadladSentences(langDir, perLangCap);
                 if (sents.isEmpty()) {
-                    System.out.printf("    %-6s: ** 0 sentences — MISSING DATA 
**%n", lang);
+                    System.out.printf(Locale.ROOT, "    %-6s: ** 0 sentences — 
MISSING DATA **%n", lang);
                 } else {
                     long totalChars = 0;
                     for (String s : sents) totalChars += s.length();
-                    System.out.printf("    %-6s: %,8d sentences  avg_len=%,.0f 
chars%n",
+                    System.out.printf(Locale.ROOT, "    %-6s: %,8d sentences  
avg_len=%,.0f chars%n",
                             lang, sents.size(), (double) totalChars / 
sents.size());
                 }
                 allSentences.addAll(sents);
@@ -632,10 +633,10 @@ public class BuildCharsetTrainingData {
                 splitBytes.put(split, totalBytes);
                 double budgetPct = 100.0 * totalBytes / budget;
                 if (ambiguousDropped > 0) {
-                    System.out.printf("    %s: %,d samples  %,d bytes (%.1f%% 
of budget)  (%,d ambiguous-dropped)%n",
+                    System.out.printf(Locale.ROOT, "    %s: %,d samples  %,d 
bytes (%.1f%% of budget)  (%,d ambiguous-dropped)%n",
                             split, written, totalBytes, budgetPct, 
ambiguousDropped);
                 } else {
-                    System.out.printf("    %s: %,d samples  %,d bytes (%.1f%% 
of budget)%n",
+                    System.out.printf(Locale.ROOT, "    %s: %,d samples  %,d 
bytes (%.1f%% of budget)%n",
                             split, written, totalBytes, budgetPct);
                 }
             }
@@ -647,14 +648,14 @@ public class BuildCharsetTrainingData {
 
         // Summary table
         System.out.println("\n=== SUMMARY ===");
-        System.out.printf("%-22s %8s %12s %8s %12s %8s %12s%n",
+        System.out.printf(Locale.ROOT, "%-22s %8s %12s %8s %12s %8s %12s%n",
                 "Charset", "Train", "Train MB", "DevTest", "DT MB", "Test", 
"Test MB");
         System.out.println("-".repeat(100));
         for (Map.Entry<String, Map<String, Integer>> e : manifest.entrySet()) {
             String cs = e.getKey();
             Map<String, Integer> sc = e.getValue();
             Map<String, Long> bt = byteTotals.getOrDefault(cs, 
Collections.emptyMap());
-            System.out.printf("%-22s %,8d %10.1f MB %,8d %10.1f MB %,8d %10.1f 
MB%n",
+            System.out.printf(Locale.ROOT, "%-22s %,8d %10.1f MB %,8d %10.1f 
MB %,8d %10.1f MB%n",
                     cs,
                     sc.getOrDefault("train", 0),
                     bt.getOrDefault("train", 0L) / 1_000_000.0,
@@ -670,7 +671,7 @@ public class BuildCharsetTrainingData {
         for (Map.Entry<String, Map<String, Integer>> e : manifest.entrySet()) {
             int train = e.getValue().getOrDefault("train", 0);
             if (train > 0 && train < 1000) {
-                System.out.printf("WARNING: %s has only %,d train samples — 
check source data!%n",
+                System.out.printf(Locale.ROOT, "WARNING: %s has only %,d train 
samples — check source data!%n",
                         e.getKey(), train);
                 anyWarnings = true;
             }
@@ -783,7 +784,7 @@ public class BuildCharsetTrainingData {
             }
             int added = result.size() - before;
             if (added > 0) {
-                System.out.printf("      (loaded %,d from %s)%n", added, 
filename);
+                System.out.printf(Locale.ROOT, "      (loaded %,d from %s)%n", 
added, filename);
             }
         }
         return result;
@@ -874,7 +875,7 @@ public class BuildCharsetTrainingData {
             stopReason = "unknown";
         }
         if (encodeRejected > 0 || !"hit byte budget".equals(stopReason)) {
-            System.out.printf("      [stop: %s | encode-rejected=%,d | "
+            System.out.printf(Locale.ROOT, "      [stop: %s | 
encode-rejected=%,d | "
                             + "sentences-consumed=%,d/%,d]%n",
                     stopReason, encodeRejected, sentIdx, sentences.size());
         }
diff --git 
a/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/DiagnoseDiscrimination.java
 
b/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/DiagnoseDiscrimination.java
index 29a16e517e..46c0ba7ee2 100644
--- 
a/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/DiagnoseDiscrimination.java
+++ 
b/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/DiagnoseDiscrimination.java
@@ -386,7 +386,7 @@ public final class DiagnoseDiscrimination {
         for (int i = 0; i < s.length(); ) {
             int cp = s.codePointAt(i);
             if (cp < 0x20 || cp == 0x7F) {
-                sb.append(String.format("\\x%02X", cp));
+                sb.append(String.format(Locale.ROOT, "\\x%02X", cp));
             } else if (cp == 0xFFFD) {
                 sb.append("<FFFD>");
             } else {
diff --git a/tika-ml/tika-ml-junkdetect-tools/pom.xml 
b/tika-ml/tika-ml-junkdetect-tools/pom.xml
index fe720dafe8..694377939f 100644
--- a/tika-ml/tika-ml-junkdetect-tools/pom.xml
+++ b/tika-ml/tika-ml-junkdetect-tools/pom.xml
@@ -83,13 +83,6 @@
         </configuration>
       </plugin>
       <!-- Tools package uses System.out/printf freely -->
-      <plugin>
-        <groupId>de.thetaphi</groupId>
-        <artifactId>forbiddenapis</artifactId>
-        <configuration>
-          <skip>true</skip>
-        </configuration>
-      </plugin>
     </plugins>
   </build>
 
diff --git 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
index 4936e70886..df26127d58 100644
--- 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
+++ 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
@@ -64,7 +64,7 @@ public final class BoundaryBigramAudit {
                     .sorted().toArray(Path[]::new);
         }
 
-        System.out.printf("%-22s %14s %14s %14s %14s %12s | %14s %14s%n",
+        System.out.printf(Locale.ROOT, "%-22s %14s %14s %14s %14s %12s | %14s 
%14s%n",
                 "script", "in_S_occ", "boundary_occ", "foreign_occ",
                 "ascii_run_occ", "total_occ",
                 "drop_foreign_dist", "drop_asciirun_dist");
@@ -135,7 +135,7 @@ public final class BoundaryBigramAudit {
             int distForeignDrop = distinctKeptUnderForeignDrop.size();
             int distAsciiDrop = distinctKeptUnderAsciiDrop.size();
 
-            System.out.printf("%-22s %,14d %,14d %,14d %,14d %,12d | %,14d 
%,14d%n",
+            System.out.printf(Locale.ROOT, "%-22s %,14d %,14d %,14d %,14d 
%,12d | %,14d %,14d%n",
                     name.toLowerCase(Locale.ROOT), inS, boundary, foreign, 
asciiRun, total,
                     distAll - distForeignDrop, distAll - distAsciiDrop);
         }
diff --git 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
index 8806be34aa..75b0a15ac1 100644
--- 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
+++ 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
@@ -159,16 +159,16 @@ public class BuildJunkTrainingData {
         System.out.println("  data-dir:               " + dataDir);
         System.out.println("  output-dir:             " + outputDir);
         System.out.println("  --- config (JunkDetectorTrainingConfig) ---");
-        System.out.printf( "  total-budget-bytes:     %,d (%.1f MB)%n",
+        System.out.printf(Locale.ROOT,  "  total-budget-bytes:     %,d (%.1f 
MB)%n",
                 totalBudgetBytes, totalBudgetBytes / 1_000_000.0);
-        System.out.printf( "  per-language-cap:       %,d (%.1f MB)%n",
+        System.out.printf(Locale.ROOT,  "  per-language-cap:       %,d (%.1f 
MB)%n",
                 perLanguageCapBytes, perLanguageCapBytes / 1_000_000.0);
-        System.out.printf( "  min-bytes:              %d%n", minBytes);
-        System.out.printf( "  max-punc-frac:          %.2f%n", maxPuncFrac);
-        System.out.printf( "  min-target-script-frac: %.2f%n", 
minTargetScriptFrac);
-        System.out.printf( "  min-dev-sentences:      %d  (min total ≈ %d)%n",
+        System.out.printf(Locale.ROOT,  "  min-bytes:              %d%n", 
minBytes);
+        System.out.printf(Locale.ROOT,  "  max-punc-frac:          %.2f%n", 
maxPuncFrac);
+        System.out.printf(Locale.ROOT,  "  min-target-script-frac: %.2f%n", 
minTargetScriptFrac);
+        System.out.printf(Locale.ROOT,  "  min-dev-sentences:      %d  (min 
total ≈ %d)%n",
                 minDevSentences, (int)(minDevSentences / DEV_FRAC));
-        System.out.printf( "  seed:                   %d%n", seed);
+        System.out.printf(Locale.ROOT,  "  seed:                   %d%n", 
seed);
         if (!dropScripts.isEmpty()) {
             System.out.println("  drop-scripts:           " + dropScripts);
         }
@@ -198,19 +198,19 @@ public class BuildJunkTrainingData {
                 String script = detectDominantScript(langDir, 
scriptSampleLines);
                 langToScript.put(lang, script);
                 scriptGroups.computeIfAbsent(script, k -> new 
ArrayList<>()).add(langDir);
-                System.out.printf("  %-12s → %s%n", lang, script);
+                System.out.printf(Locale.ROOT, "  %-12s → %s%n", lang, script);
             }
         }
 
         if (!dropScripts.isEmpty()) {
             for (String s : dropScripts) {
                 if (scriptGroups.remove(s) != null) {
-                    System.out.printf("  DROP script: %s%n", s);
+                    System.out.printf(Locale.ROOT, "  DROP script: %s%n", s);
                 }
             }
         }
 
-        System.out.printf("%n  → %d languages, %d script groups%n",
+        System.out.printf(Locale.ROOT, "%n  → %d languages, %d script 
groups%n",
                 langToScript.size(), scriptGroups.size());
 
         // 
-----------------------------------------------------------------------
@@ -233,7 +233,7 @@ public class BuildJunkTrainingData {
 
             double entropy = computeBigramEntropy(sample);
             scriptEntropy.put(script, entropy);
-            System.out.printf("  %-20s H=%.3f bits  (%d sentences)%n",
+            System.out.printf(Locale.ROOT, "  %-20s H=%.3f bits  (%d 
sentences)%n",
                     script, entropy, sample.size());
         }
 
@@ -251,13 +251,13 @@ public class BuildJunkTrainingData {
             long budget = (long) (totalBudgetBytes * e.getValue() / 
totalEntropy);
             Long override = scriptBudgetOverrides.get(e.getKey());
             if (override != null) {
-                System.out.printf("  %-20s H=%.3f → %,d bytes (%.1f MB)"
+                System.out.printf(Locale.ROOT, "  %-20s H=%.3f → %,d bytes 
(%.1f MB)"
                         + "  [OVERRIDE: was %,d (%.1f MB)]%n",
                         e.getKey(), e.getValue(), override, override / 
1_000_000.0,
                         budget, budget / 1_000_000.0);
                 budget = override;
             } else {
-                System.out.printf("  %-20s H=%.3f → %,d bytes (%.1f MB)%n",
+                System.out.printf(Locale.ROOT, "  %-20s H=%.3f → %,d bytes 
(%.1f MB)%n",
                         e.getKey(), e.getValue(), budget, budget / 
1_000_000.0);
             }
             scriptBudget.put(e.getKey(), budget);
@@ -265,7 +265,7 @@ public class BuildJunkTrainingData {
         // Warn about overrides for scripts that aren't in the bucket set.
         for (String k : scriptBudgetOverrides.keySet()) {
             if (!scriptBudget.containsKey(k)) {
-                System.err.printf("WARNING: --script-budget-override for %s 
ignored"
+                System.err.printf(Locale.ROOT, "WARNING: 
--script-budget-override for %s ignored"
                         + " (script not in bucket set)%n", k);
             }
         }
@@ -315,7 +315,7 @@ public class BuildJunkTrainingData {
                         sentences);
                 totalBytesLoaded += langBytes;
                 if (langBytes > 0) {
-                    System.out.printf("  %-12s %-20s +%,d bytes%n",
+                    System.out.printf(Locale.ROOT, "  %-12s %-20s +%,d 
bytes%n",
                             script, langDir.getFileName(), langBytes);
                 }
             }
@@ -335,7 +335,7 @@ public class BuildJunkTrainingData {
 
         // Round 2: redistribute surplus to saturated scripts proportional to 
entropy
         if (surplus > 0) {
-            System.out.printf(
+            System.out.printf(Locale.ROOT, 
                     "\n--- Phase 4b: Redistributing %,d surplus bytes (%.1f 
MB) ---\n",
                     surplus, surplus / 1_000_000.0);
 
@@ -377,7 +377,7 @@ public class BuildJunkTrainingData {
                 if (!sentences.isEmpty()) {
                     allSentences.put(script, sentences);
                     actualBytes.put(script, totalBytesLoaded);
-                    System.out.printf("  %-20s +%,d extra → %,d total bytes, 
%,d sentences%n",
+                    System.out.printf(Locale.ROOT, "  %-20s +%,d extra → %,d 
total bytes, %,d sentences%n",
                             script, extra, totalBytesLoaded, sentences.size());
                 }
             }
@@ -395,7 +395,7 @@ public class BuildJunkTrainingData {
 
             int expectedDevSize = (int) (sentences.size() * DEV_FRAC);
             if (sentences.isEmpty() || expectedDevSize < minDevSentences) {
-                System.out.printf(
+                System.out.printf(Locale.ROOT, 
                         "  SKIP %-20s — %,d sentences → dev=%d < 
min-dev-sentences=%d%n",
                         script, sentences.size(), expectedDevSize, 
minDevSentences);
                 manifestStats.put(script, new long[]{0, 0, 0, 0, 0});
@@ -418,7 +418,7 @@ public class BuildJunkTrainingData {
             long totalBytesLoaded = actualBytes.getOrDefault(script, 0L);
             manifestStats.put(script,
                     new long[]{totalBytesLoaded, sentences.size(), nTrain, 
nDev, test.size()});
-            System.out.printf(
+            System.out.printf(Locale.ROOT, 
                     "  WROTE %-12s — %,d bytes, %,d sentences (train=%,d 
dev=%,d test=%,d)%n",
                     script, totalBytesLoaded, sentences.size(),
                     nTrain, nDev, test.size());
@@ -440,7 +440,7 @@ public class BuildJunkTrainingData {
                 String langs = scriptGroups.get(script).stream()
                         .map(p -> p.getFileName().toString())
                         .reduce((a, b) -> a + "," + b).orElse("");
-                w.write(String.format("%s\t%.3f\t%d\t%d\t%d\t%d\t%d\t%d\t%s%n",
+                w.write(String.format(Locale.ROOT, 
"%s\t%.3f\t%d\t%d\t%d\t%d\t%d\t%d\t%s%n",
                         script, entropy, budget,
                         stats[0], stats[1], stats[2], stats[3], stats[4], 
langs));
             }
diff --git 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/DebugScriptRuns.java
 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/DebugScriptRuns.java
index 36f3a897a0..8044c3b576 100644
--- 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/DebugScriptRuns.java
+++ 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/DebugScriptRuns.java
@@ -25,6 +25,7 @@ import java.nio.file.Paths;
 import java.util.ArrayList;
 import java.util.Arrays;
 import java.util.List;
+import java.util.Locale;
 import java.util.Map;
 import java.util.TreeMap;
 import java.util.regex.Matcher;
@@ -139,7 +140,7 @@ public class DebugScriptRuns {
         System.out.println("Script roll-up (script: cps, utf8_bytes, runs, 
modeled):");
         for (Map.Entry<String, int[]> e : totals.entrySet()) {
             int[] v = e.getValue();
-            System.out.printf("  %-15s cps=%-5d bytes=%-6d runs=%-4d 
modeled=%s%n",
+            System.out.printf(Locale.ROOT, "  %-15s cps=%-5d bytes=%-6d 
runs=%-4d modeled=%s%n",
                     e.getKey(), v[0], v[1], v[2], v[3] == 1 ? "Y" : "N");
         }
         System.out.println();
@@ -166,7 +167,7 @@ public class DebugScriptRuns {
         TextQualityScore score = detector.score(decoded);
         System.out.println("  detector.score() z:    "
                 + (score.isUnknown() ? "UNKNOWN(" + score.getDominantScript() 
+ ")"
-                : String.format("%.3f (script=%s)", score.getZScore(), 
score.getDominantScript())));
+                : String.format(Locale.ROOT, "%.3f (script=%s)", 
score.getZScore(), score.getDominantScript())));
 
         // Print the longest 10 runs so we can see what's actually in there.
         System.out.println();
@@ -178,7 +179,7 @@ public class DebugScriptRuns {
             String preview = r.text.length() > 30
                     ? r.text.substring(0, 30) + "…" : r.text;
             preview = preview.replace("\n", "\\n").replace("\r", "\\r");
-            System.out.printf("  %-15s cps=%-4d bytes=%-4d preview=%s%n",
+            System.out.printf(Locale.ROOT, "  %-15s cps=%-4d bytes=%-4d 
preview=%s%n",
                     r.script, r.text.codePointCount(0, r.text.length()), 
u.length, preview);
         }
     }
diff --git 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
index 21cdefa954..ba111cd3c2 100644
--- 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
+++ 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
@@ -68,7 +68,7 @@ public final class LineScriptFractions {
             System.exit(1);
         }
 
-        System.out.printf("%-20s %10s %10s | %s%n",
+        System.out.printf(Locale.ROOT, "%-20s %10s %10s | %s%n",
                 "script", "lines", "<5%",
                 "lines at target-frac threshold (cumulative dropped %)");
         System.out.println("                                            "
@@ -81,7 +81,7 @@ public final class LineScriptFractions {
                     .toUpperCase(Locale.ROOT);
             Character.UnicodeScript target = mapScript(name);
             if (target == null) {
-                System.out.printf("%-20s  (no UnicodeScript mapping for 
'%s')%n", name, name);
+                System.out.printf(Locale.ROOT, "%-20s  (no UnicodeScript 
mapping for '%s')%n", name, name);
                 continue;
             }
 
@@ -131,11 +131,11 @@ public final class LineScriptFractions {
                     if (hi <= t) dropped += bucketCounts[j];
                 }
                 double pct = 100.0 * dropped / Math.max(1, lines);
-                sb.append(String.format(" %6.1f", pct));
+                sb.append(String.format(Locale.ROOT, " %6.1f", pct));
             }
 
             long below5 = bucketCounts[0];
-            System.out.printf("%-20s %,10d %,10d |%s%n",
+            System.out.printf(Locale.ROOT, "%-20s %,10d %,10d |%s%n",
                     name.toLowerCase(Locale.ROOT), lines, below5, 
sb.toString());
         }
     }
diff --git 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/ScriptCensus.java
 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/ScriptCensus.java
index b384d5f4c5..c7a0421113 100644
--- 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/ScriptCensus.java
+++ 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/ScriptCensus.java
@@ -27,6 +27,7 @@ import java.util.ArrayList;
 import java.util.Comparator;
 import java.util.HashMap;
 import java.util.List;
+import java.util.Locale;
 import java.util.Map;
 import java.util.zip.GZIPInputStream;
 
@@ -116,8 +117,8 @@ public final class ScriptCensus {
             }
         }
 
-        System.out.printf("File: %s%n", file);
-        System.out.printf("  lines sampled: %,d   total codepoints (excl. 
COMMON/INHERITED): %,d%n%n",
+        System.out.printf(Locale.ROOT, "File: %s%n", file);
+        System.out.printf(Locale.ROOT, "  lines sampled: %,d   total 
codepoints (excl. COMMON/INHERITED): %,d%n%n",
                 lines, total);
 
         if (total == 0) {
@@ -135,7 +136,7 @@ public final class ScriptCensus {
             double pct = 100.0 * c / total;
             double cumPct = 100.0 * cumulative / total;
             if (pct < 0.01 && c < 100) continue;
-            System.out.printf("    %-22s %,14d  %6.2f%%  (cum %6.2f%%)%n",
+            System.out.printf(Locale.ROOT, "    %-22s %,14d  %6.2f%%  (cum 
%6.2f%%)%n",
                     e.getKey(), c, pct, cumPct);
         }
 
@@ -149,7 +150,7 @@ public final class ScriptCensus {
             long c = e.getValue()[0];
             double pct = 100.0 * c / domTotal;
             if (pct < 0.05) continue;
-            System.out.printf("    %-22s %,12d  %6.2f%% of lines%n",
+            System.out.printf(Locale.ROOT, "    %-22s %,12d  %6.2f%% of 
lines%n",
                     e.getKey(), c, pct);
         }
     }
diff --git 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
index 4bebad6fa3..33629e97f2 100644
--- 
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
+++ 
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
@@ -238,11 +238,11 @@ public class TrainJunkModel {
         System.out.println("  data-dir:           " + dataDir);
         System.out.println("  output:             " + output);
         System.out.println("  --- format constants (TrainJunkModel) ---");
-        System.out.printf( "  backoff_alpha:      %.2f%n", BACKOFF_ALPHA);
+        System.out.printf(Locale.ROOT,  "  backoff_alpha:      %.2f%n", 
BACKOFF_ALPHA);
         System.out.println("  --- config (JunkDetectorTrainingConfig) ---");
-        System.out.printf( "  min_bigram_count:   %d%n", minBigramCount);
-        System.out.printf( "  oa_load_factor:     %.2f%n", loadFactor);
-        System.out.printf( "  key_index_bits:     %d%n", keyIndexBits);
+        System.out.printf(Locale.ROOT,  "  min_bigram_count:   %d%n", 
minBigramCount);
+        System.out.printf(Locale.ROOT,  "  oa_load_factor:     %.2f%n", 
loadFactor);
+        System.out.printf(Locale.ROOT,  "  key_index_bits:     %d%n", 
keyIndexBits);
 
         if (!Files.isDirectory(dataDir)) {
             System.err.println("ERROR: data-dir not found: " + dataDir);
@@ -250,7 +250,7 @@ public class TrainJunkModel {
         }
 
         int blockN = 
org.apache.tika.ml.junkdetect.UnicodeBlockRanges.bucketCount();
-        System.out.printf("Block bucketing: %d named blocks + 1 unassigned "
+        System.out.printf(Locale.ROOT, "Block bucketing: %d named blocks + 1 
unassigned "
                 + "(scheme version %d, JVM-independent)%n",
                 blockN - 1, 
org.apache.tika.ml.junkdetect.UnicodeBlockRanges.SCHEME_VERSION);
         long t0 = System.currentTimeMillis();
@@ -303,7 +303,7 @@ public class TrainJunkModel {
                     unigramsByScript.get(script), totals[0],
                     minBigramCount, loadFactor, keyIndexBits);
             f1TablesByScript.put(script, tables);
-            System.out.printf("  [%s] %s (%dms)%n", script, 
tables.statsString(),
+            System.out.printf(Locale.ROOT, "  [%s] %s (%dms)%n", script, 
tables.statsString(),
                     System.currentTimeMillis() - t0);
         }
 
@@ -331,7 +331,7 @@ public class TrainJunkModel {
             float[] cal = scores.isEmpty() ? new float[]{0f, 1f} : 
muSigma(scores);
             cal[1] = Math.max(cal[1], Z1_MIN_SIGMA);   // floor degenerate 
(under-trained) sigma
             f1Calibrations.put(script, cal);
-            System.out.printf("  [%s] mu=%.4f sigma=%.4f (%,d windows)%n",
+            System.out.printf(Locale.ROOT, "  [%s] mu=%.4f sigma=%.4f (%,d 
windows)%n",
                     script, cal[0], cal[1], scores.size());
         }
 
@@ -350,15 +350,15 @@ public class TrainJunkModel {
         scriptTransTable = quantizeDequantizeRoundTrip(scriptTransTable);
         float[] scriptTransCal = calibrateScriptTransitions(allTrainFiles, 
scriptTransTable,
                 scriptBucketMap, numScriptBuckets);
-        System.out.printf("  scriptTrans: mu=%.4f sigma=%.4f%n",
+        System.out.printf(Locale.ROOT, "  scriptTrans: mu=%.4f sigma=%.4f%n",
                 scriptTransCal[0], scriptTransCal[1]);
 
         float[] blockTable = 
quantizeDequantizeRoundTrip(trainGlobalBlockTable(allTrainFiles));
         float[] blockCal = computeGlobalBlockCalibration(allTrainFiles, 
blockTable);
-        System.out.printf("  block:       mu=%.4f sigma=%.4f%n", blockCal[0], 
blockCal[1]);
+        System.out.printf(Locale.ROOT, "  block:       mu=%.4f sigma=%.4f%n", 
blockCal[0], blockCal[1]);
 
         float[] controlCal = computeGlobalControlCalibration(allTrainFiles);
-        System.out.printf("  control:     mu=%.6f sigma=%.6f%n", 
controlCal[0], controlCal[1]);
+        System.out.printf(Locale.ROOT, "  control:     mu=%.6f sigma=%.6f%n", 
controlCal[0], controlCal[1]);
 
         // 
-----------------------------------------------------------------------
         // Phase 3 — ONE global combiner over z1..z9, trained pointwise
@@ -376,17 +376,17 @@ public class TrainJunkModel {
 
         t0 = System.currentTimeMillis();
         float[] combiner = trainGlobalCombiner(featExtractor, trainFilePaths);
-        System.out.printf(
+        System.out.printf(Locale.ROOT, 
                 "  global w=[%.3f,%.3f,%.3f,%.3f,%.3f,%.3f,%.3f,%.3f,%.3f] 
bias=%.3f (%dms)%n",
                 combiner[0], combiner[1], combiner[2], combiner[3], 
combiner[4], combiner[5],
                 combiner[6], combiner[7], combiner[8], combiner[9],
                 System.currentTimeMillis() - t0);
 
-        System.out.printf("%nWriting model (%d scripts, blockN=%d, 
scriptBuckets=%d) → %s%n",
+        System.out.printf(Locale.ROOT, "%nWriting model (%d scripts, 
blockN=%d, scriptBuckets=%d) → %s%n",
                 f1TablesByScript.size(), blockN, numScriptBuckets, output);
         saveModel(f1TablesByScript, f1Calibrations, blockTable, blockCal, 
controlCal,
                 combiner, scriptBuckets, scriptTransTable, scriptTransCal, 
output);
-        System.out.printf("Model size: %,d bytes (%.1f KB)%n",
+        System.out.printf(Locale.ROOT, "Model size: %,d bytes (%.1f KB)%n",
                 Files.size(output), Files.size(output) / 1024.0);
         System.out.println("Done.");
     }
@@ -1081,7 +1081,7 @@ public class TrainJunkModel {
             values[i] = (byte) (sortable[i] & 0xFF);
         }
 
-        System.out.printf(
+        System.out.printf(Locale.ROOT, 
                 "    pair_counts: distinct=%,d, kept=%,d (>=%d), dropped=%,d  "
                 + "cp_index=%,d  bigram_entries=%,d%n",
                 totalDistinct, keptPairs, minBigramCount, dropped,
@@ -1218,7 +1218,7 @@ public class TrainJunkModel {
                 }
             }
         }
-        System.out.printf("  examples: good=%,d bad=%,d pairs=%,d%n",
+        System.out.printf(Locale.ROOT, "  examples: good=%,d bad=%,d 
pairs=%,d%n",
                 good.size(), bad.size(), pairCorrect.size());
         return fitContrastiveCombiner(good, bad, pairCorrect, pairWrong);
     }
@@ -1612,7 +1612,7 @@ public class TrainJunkModel {
                 }
             }
         }
-        System.out.printf("%,d script transitions across %d files%n", 
totalTransitions, trainFiles.size());
+        System.out.printf(Locale.ROOT, "%,d script transitions across %d 
files%n", totalTransitions, trainFiles.size());
         return laplaceSmoothLogProb(counts, numBuckets);
     }
 
@@ -1638,7 +1638,7 @@ public class TrainJunkModel {
                 }
             }
         }
-        System.out.printf("%,d dev windows pooled%n", scores.size());
+        System.out.printf(Locale.ROOT, "%,d dev windows pooled%n", 
scores.size());
         return muSigma(scores);
     }
 
diff --git 
a/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
 
b/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
index c135f9357e..d6f1ed5a82 100644
--- 
a/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
+++ 
b/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
@@ -271,14 +271,14 @@ public final class BuildJunkAugmentationData {
                 ? null
                 : loadProfileCsv(profileCsv);
         if (profiles != null) {
-            System.out.printf("  loaded %,d profile rows%n", profiles.size());
+            System.out.printf(Locale.ROOT, "  loaded %,d profile rows%n", 
profiles.size());
         }
 
         // --- Phase 1: discover baseline scripts + line counts 
-------------------
         System.out.println("\n--- Phase 1: scanning baseline train files ---");
         Map<String, Long> baselineLineCounts = 
scanBaselineLineCounts(baselineDir);
         for (Map.Entry<String, Long> e : baselineLineCounts.entrySet()) {
-            System.out.printf("  %-20s baseline=%,d lines%n", e.getKey(), 
e.getValue());
+            System.out.printf(Locale.ROOT, "  %-20s baseline=%,d lines%n", 
e.getKey(), e.getValue());
         }
 
         // --- Phase 2: walk extracts 
---------------------------------------------
@@ -369,17 +369,17 @@ public final class BuildJunkAugmentationData {
             }
         }
 
-        System.out.printf("  total extracts seen:    %,d%n", totalSeen);
-        System.out.printf("  dropped no-content:     %,d%n", droppedNoContent);
-        System.out.printf("  dropped short:          %,d%n", droppedShort);
+        System.out.printf(Locale.ROOT, "  total extracts seen:    %,d%n", 
totalSeen);
+        System.out.printf(Locale.ROOT, "  dropped no-content:     %,d%n", 
droppedNoContent);
+        System.out.printf(Locale.ROOT, "  dropped short:          %,d%n", 
droppedShort);
         if (profiles != null) {
-            System.out.printf("  dropped no-profile:     %,d%n", 
droppedNoProfile);
-            System.out.printf("  dropped OOV>%.2f:       %,d%n", maxOov, 
droppedOov);
-            System.out.printf("  dropped langness<%.2f:  %,d%n", minLangness, 
droppedLangness);
+            System.out.printf(Locale.ROOT, "  dropped no-profile:     %,d%n", 
droppedNoProfile);
+            System.out.printf(Locale.ROOT, "  dropped OOV>%.2f:       %,d%n", 
maxOov, droppedOov);
+            System.out.printf(Locale.ROOT, "  dropped langness<%.2f:  %,d%n", 
minLangness, droppedLangness);
         }
-        System.out.printf("  dropped mixed-script:   %,d%n", 
droppedMixedScript);
-        System.out.printf("  dropped no-baseline:    %,d%n", 
droppedNoBaseline);
-        System.out.printf("  contributed ≥1 chunk:   %,d%n", accepted);
+        System.out.printf(Locale.ROOT, "  dropped mixed-script:   %,d%n", 
droppedMixedScript);
+        System.out.printf(Locale.ROOT, "  dropped no-baseline:    %,d%n", 
droppedNoBaseline);
+        System.out.printf(Locale.ROOT, "  contributed ≥1 chunk:   %,d%n", 
accepted);
 
         // --- Phase 3: apply per-script gates and caps 
---------------------------
         System.out.println("\n--- Phase 3: per-script gating and capping ---");
@@ -398,7 +398,7 @@ public final class BuildJunkAugmentationData {
             long cap = Math.min(hardCap, fracCapVal);
 
             if (docs < minDocs) {
-                System.out.printf(
+                System.out.printf(Locale.ROOT, 
                         "  SKIP %-20s docs=%-6d (<%d gate)  chunks=%,d  
cap=%,d%n",
                         script, docs, minDocs, chunks.size(), cap);
                 manifest.put(script, new long[]{docs, chunks.size(), 0, 
baselineLines, cap});
@@ -428,7 +428,7 @@ public final class BuildJunkAugmentationData {
                 for (int k = quota; kept.size() < cap && k < withSym.size(); 
k++) {
                     kept.add(withSym.get(k));
                 }
-                System.out.printf(
+                System.out.printf(Locale.ROOT, 
                         "  KEEP %-20s docs=%-6d chunks=%,8d -> append=%,6d  "
                         + "(baseline=%,d, cap=%,d, symbol-quota=%d, 
symbol-bearing-pool=%d)%n",
                         script, docs, chunks.size(), kept.size(), 
baselineLines, cap,
@@ -437,7 +437,7 @@ public final class BuildJunkAugmentationData {
                 kept = chunks.size() > cap
                         ? new ArrayList<>(chunks.subList(0, (int) cap))
                         : chunks;
-                System.out.printf(
+                System.out.printf(Locale.ROOT, 
                         "  KEEP %-20s docs=%-6d chunks=%,8d -> append=%,6d  
(baseline=%,d, cap=%,d)%n",
                         script, docs, chunks.size(), kept.size(), 
baselineLines, cap);
             }
@@ -468,10 +468,10 @@ public final class BuildJunkAugmentationData {
                     List<String> add = finalLines.get(script);
                     if (add != null && !add.isEmpty()) {
                         rewriteTrainWithAppend(src, dst, add);
-                        System.out.printf("  WROTE  %-30s +%,d lines 
appended%n", name, add.size());
+                        System.out.printf(Locale.ROOT, "  WROTE  %-30s +%,d 
lines appended%n", name, add.size());
                     } else {
                         Files.copy(src, dst);
-                        System.out.printf("  COPY   %-30s (no 
augmentation)%n", name);
+                        System.out.printf(Locale.ROOT, "  COPY   %-30s (no 
augmentation)%n", name);
                     }
                 } else {
                     Files.copy(src, dst);
@@ -485,7 +485,7 @@ public final class BuildJunkAugmentationData {
             
w.write("script\tdocs\tchunks_pre_cap\tlines_appended\tbaseline_lines\tcap\n");
             for (Map.Entry<String, long[]> e : manifest.entrySet()) {
                 long[] r = e.getValue();
-                w.write(String.format("%s\t%d\t%d\t%d\t%d\t%d%n",
+                w.write(String.format(Locale.ROOT, "%s\t%d\t%d\t%d\t%d\t%d%n",
                         e.getKey(), r[0], r[1], r[2], r[3], r[4]));
             }
         }
diff --git a/tika-ml/tika-ml-junkdetect/pom.xml 
b/tika-ml/tika-ml-junkdetect/pom.xml
index af9e7c456f..ae30c8247d 100644
--- a/tika-ml/tika-ml-junkdetect/pom.xml
+++ b/tika-ml/tika-ml-junkdetect/pom.xml
@@ -97,13 +97,6 @@
         </configuration>
       </plugin>
       <!-- Diagnostic tests print to stdout / use default-locale formatting 
freely. -->
-      <plugin>
-        <groupId>de.thetaphi</groupId>
-        <artifactId>forbiddenapis</artifactId>
-        <configuration>
-          <skip>true</skip>
-        </configuration>
-      </plugin>
     </plugins>
   </build>
 
diff --git 
a/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/LatinSiblingComparisonTest.java
 
b/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/LatinSiblingComparisonTest.java
index 8965d841fe..57b7fb2632 100644
--- 
a/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/LatinSiblingComparisonTest.java
+++ 
b/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/LatinSiblingComparisonTest.java
@@ -21,6 +21,7 @@ import static org.junit.jupiter.api.Assertions.assertEquals;
 import java.nio.charset.Charset;
 import java.util.ArrayList;
 import java.util.List;
+import java.util.Locale;
 
 import org.junit.jupiter.api.BeforeAll;
 import org.junit.jupiter.api.Test;
@@ -108,11 +109,11 @@ public class LatinSiblingComparisonTest {
                 TextQualityComparison cmp = detector.compare(
                         "windows-1252", asWin1252, wrong, asWrong);
 
-                String tag = String.format("%-20s vs %-12s", probe.name, 
wrong);
+                String tag = String.format(Locale.ROOT, "%-20s vs %-12s", 
probe.name, wrong);
                 if ("windows-1252".equals(cmp.winner())) {
-                    passes.add(String.format("PASS %s  delta=%.3f", tag, 
cmp.delta()));
+                    passes.add(String.format(Locale.ROOT, "PASS %s  
delta=%.3f", tag, cmp.delta()));
                 } else {
-                    failures.add(String.format("FAIL %s  winner=%-12s 
delta=%.3f",
+                    failures.add(String.format(Locale.ROOT, "FAIL %s  
winner=%-12s delta=%.3f",
                             tag, cmp.winner(), cmp.delta()));
                 }
             }
@@ -121,7 +122,7 @@ public class LatinSiblingComparisonTest {
         System.out.println("\n=== Latin SBCS sibling comparison: " + label + " 
===");
         passes.forEach(System.out::println);
         failures.forEach(System.out::println);
-        System.out.printf("%d pass, %d fail (of %d cells)%n",
+        System.out.printf(Locale.ROOT, "%d pass, %d fail (of %d cells)%n",
                 passes.size(), failures.size(), passes.size() + 
failures.size());
 
         assertEquals(0, failures.size(),
diff --git a/tika-parent/checkstyle.xml b/tika-parent/checkstyle.xml
index cf9928edf4..72c5050e1f 100644
--- a/tika-parent/checkstyle.xml
+++ b/tika-parent/checkstyle.xml
@@ -65,8 +65,8 @@
     <!--<module name="FileContentsHolder"/>-->
     <module name="IllegalImport">
       <property name="regexp" value="true"/>
-      <!-- Reject any org.junit import that's not also org.junit.jupiter: -->
-      <property name="illegalClasses" value="^org\.junit\.(?!jupiter\.).+"/>
+      <!-- Reject JUnit 4: any org.junit import outside org.junit.jupiter and 
org.junit.platform -->
+      <property name="illegalClasses" 
value="^org\.junit\.(?!(jupiter|platform)\.).+"/>
     </module>
     <module name="OuterTypeFilename"/>
     <module name="IllegalTokenText">
diff --git a/tika-parent/pom.xml b/tika-parent/pom.xml
index 4e6e363b54..edd2d7335c 100644
--- a/tika-parent/pom.xml
+++ b/tika-parent/pom.xml
@@ -1522,8 +1522,9 @@
         <artifactId>maven-surefire-plugin</artifactId>
         <version>${maven.surefire.version}</version>
         <configuration>
-          <!-- for manual testing of i18n, try for example: -Duser.language=zh 
-Duser.region=CN or
-          -Duser.language=de -Duser.country=DE -->
+          <!-- -Dtika.test.locale=random (CI) or =tr-TR runs the tests under 
that default locale
+               (RandomLocaleListener). -Duser.language here would not work: 
surefire sets
+               it in the fork only after the default Locale is fixed. -->
           <!-- java.io.tmpdir MUST be quoted: argLine is split on whitespace, 
and the Windows
                job deliberately checks out into a path containing a space. It 
must also be a
                real -D on the command line, not <systemPropertyVariables>: 
Files.createTempFile
@@ -1734,6 +1735,28 @@
   </build>
 
   <profiles>
+    <profile>
+      <!-- Puts RandomLocaleListener on the test classpath of every module 
that has tests:
+           each test JVM then runs under a random default locale. A profile, 
not a plain
+           dependency: aggregator poms and tika-test-support itself (which 
keeps its test
+           in tika-core for this reason) must not depend on it, or the reactor 
cycles. -->
+      <id>test-support</id>
+      <activation>
+        <file>
+          <exists>${basedir}/src/test/java</exists>
+        </file>
+      </activation>
+      <dependencies>
+        <dependency>
+          <groupId>org.apache.tika</groupId>
+          <artifactId>tika-test-support</artifactId>
+          <!-- revision, not project.version: flatten writes the literal Tika 
version
+               into the published tika-parent, so a foreign child of it 
resolves the right jar -->
+          <version>${revision}</version>
+          <scope>test</scope>
+        </dependency>
+      </dependencies>
+    </profile>
     <profile>
       <id>pedantic</id>
       <build>
diff --git 
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/grib/GribParserTest.java
 
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/grib/GribParserTest.java
index 7768eda1fc..57d5806874 100644
--- 
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/grib/GribParserTest.java
+++ 
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/grib/GribParserTest.java
@@ -18,6 +18,10 @@ package org.apache.tika.parser.grib;
 
 import static org.junit.jupiter.api.Assertions.assertNotNull;
 import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.junit.jupiter.api.Assumptions.assumeTrue;
+
+import java.text.DecimalFormatSymbols;
+import java.util.Locale;
 
 import org.junit.jupiter.api.Test;
 import org.xml.sax.ContentHandler;
@@ -37,6 +41,10 @@ public class GribParserTest {
     @Test
     public void testParseGlobalMetadata() throws Exception {
         Parser parser = new GribParser();
+        // TODO: netcdf-java formats with the default locale, so dimension 
sizes and a WMO
+        // code-table file name come out in the locale's digits; report 
upstream
+        
assumeTrue(DecimalFormatSymbols.getInstance(Locale.getDefault()).getZeroDigit() 
== '0',
+                "netcdf-java needs ASCII digits in the default locale");
         Metadata metadata = new Metadata();
         ContentHandler handler = new BodyContentHandler();
         try (TikaInputStream tis = TikaInputStream.get(GribParser.class
diff --git 
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/netcdf/NetCDFParserTest.java
 
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/netcdf/NetCDFParserTest.java
index 84b8fb0ffb..0e031dce37 100644
--- 
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/netcdf/NetCDFParserTest.java
+++ 
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/netcdf/NetCDFParserTest.java
@@ -18,6 +18,10 @@ package org.apache.tika.parser.netcdf;
 
 import static org.apache.tika.TikaTest.assertContains;
 import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assumptions.assumeTrue;
+
+import java.text.DecimalFormatSymbols;
+import java.util.Locale;
 
 import org.junit.jupiter.api.Test;
 import org.xml.sax.ContentHandler;
@@ -38,6 +42,10 @@ public class NetCDFParserTest {
     @Test
     public void testParseGlobalMetadata() throws Exception {
         Parser parser = new NetCDFParser();
+        // TODO: netcdf-java formats with the default locale, so dimension 
sizes and a WMO
+        // code-table file name come out in the locale's digits; report 
upstream
+        
assumeTrue(DecimalFormatSymbols.getInstance(Locale.getDefault()).getZeroDigit() 
== '0',
+                "netcdf-java needs ASCII digits in the default locale");
         ContentHandler handler = new BodyContentHandler();
         Metadata metadata = new Metadata();
 
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
index 9434b43971..f3ee0a83a5 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
@@ -258,8 +258,8 @@ public class TesseractOCRParserTest extends TikaTest {
         assertEquals("75", m.get(TIFF.IMAGE_LENGTH));
     }
 
-    // TODO TIKA-4923: metadata-extractor lowercases the resolution unit in 
the default locale
-    @DisabledIfSystemProperty(named = "user.language", matches = "tr")
+    // TODO TIKA-4923: metadata-extractor lowercases the resolution unit in 
the default locale (tr/az dotless i)
+    @DisabledIfSystemProperty(named = "tika.test.locale", matches = 
"(tr|az)(-.*)?")
     @Test
     public void getNormalMetadataTooUnknownField() throws Exception {
         Metadata m = getXML("testTIFF.tif").metadata;
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/sas/SAS7BDATParser.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/sas/SAS7BDATParser.java
index dfa4a6099c..79c3954719 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/sas/SAS7BDATParser.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/sas/SAS7BDATParser.java
@@ -20,6 +20,7 @@ import java.io.IOException;
 import java.text.Format;
 import java.util.Collections;
 import java.util.HashMap;
+import java.util.Locale;
 import java.util.Map;
 import java.util.Set;
 
@@ -141,7 +142,8 @@ public class SAS7BDATParser implements Parser {
         Object[] row = null;
         while ((row = sas.readNext()) != null) {
             xhtml.startElement("tr");
-            for (String val : DataWriterUtil.getRowValues(sas.getColumns(), 
row, formatMap)) {
+            // parso otherwise formats dates and percents with the JVM default 
locale
+            for (String val : DataWriterUtil.getRowValues(sas.getColumns(), 
row, Locale.US, formatMap)) {
                 // Use explicit start/end, rather than element, to 
                 //  ensure that empty cells still get output
                 xhtml.startElement("td");
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/test/java/org/apache/tika/parser/sas/SAS7BDATParserTest.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/test/java/org/apache/tika/parser/sas/SAS7BDATParserTest.java
index 22b6d47635..50cb9af7e0 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/test/java/org/apache/tika/parser/sas/SAS7BDATParserTest.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/test/java/org/apache/tika/parser/sas/SAS7BDATParserTest.java
@@ -18,9 +18,7 @@ package org.apache.tika.parser.sas;
 
 import static org.junit.jupiter.api.Assertions.assertEquals;
 
-import java.text.DateFormatSymbols;
 import java.util.Arrays;
-import java.util.Locale;
 
 import org.junit.jupiter.api.Test;
 import org.xml.sax.ContentHandler;
@@ -39,8 +37,6 @@ import org.apache.tika.parser.Parser;
 import org.apache.tika.sax.BodyContentHandler;
 
 public class SAS7BDATParserTest extends TikaTest {
-    private static final String[] SHORT_MONTHS =
-            new DateFormatSymbols(Locale.getDefault()).getShortMonths();
     private Parser parser = new SAS7BDATParser();
 
     @Test
@@ -111,7 +107,7 @@ public class SAS7BDATParserTest extends TikaTest {
         assertContains("2\t4\tThis", content);
         assertContains("4\t16\tThis", content);
         assertContains("\t01-01-1960\t", content);
-        assertContains("\t01" + SHORT_MONTHS[0] + "1960:00:00", content);
+        assertContains("\t01Jan1960:00:00", content);
     }
 
     @Test
@@ -143,6 +139,6 @@ public class SAS7BDATParserTest extends TikaTest {
         assertContains("<th title=\"date\">date</th>", xml);
         // Check formatting of dates
         assertContains("<td>01-01-1960</td>", xml);
-        assertContains("<td>01" + SHORT_MONTHS[0] + "1960:00:00:10.00</td>", 
xml);
+        assertContains("<td>01Jan1960:00:00:10.00</td>", xml);
     }
 }
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/main/java/org/apache/tika/parser/image/ImageMetadataExtractor.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/main/java/org/apache/tika/parser/image/ImageMetadataExtractor.java
index 0ee7e82fbc..74be1c0378 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/main/java/org/apache/tika/parser/image/ImageMetadataExtractor.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/main/java/org/apache/tika/parser/image/ImageMetadataExtractor.java
@@ -24,6 +24,10 @@ import java.nio.channels.SeekableByteChannel;
 import java.text.DecimalFormat;
 import java.text.DecimalFormatSymbols;
 import java.text.SimpleDateFormat;
+import java.time.DateTimeException;
+import java.time.LocalDate;
+import java.time.LocalTime;
+import java.time.ZoneOffset;
 import java.util.Arrays;
 import java.util.Date;
 import java.util.Iterator;
@@ -553,13 +557,20 @@ public class ImageMetadataExtractor {
 
     static class ExifHandler implements DirectoryHandler {
         // There's a new ExifHandler for each file processed, so this is 
thread safe
-        // EXIF dates have no zone: read and write in GMT (metadata-extractor 
>= 2.20 uses the JVM zone)
+        // EXIF dates have no zone: write them in GMT
         private static final TimeZone GMT = TimeZone.getTimeZone("GMT");
         private final SimpleDateFormat dateUnspecifiedTz = 
getUnspecifiedTzDateFormat();
 
-        // metadata-extractor turns junk like "2" into year 1
-        private static Date inBounds(Date d) {
-            return d != null && TikaDates.inYearBounds(d.toInstant()) ? d : 
null;
+        // metadata-extractor parses with the default locale's calendar; 
TikaDates does not
+        private static Date exifDate(Directory directory, int tag) {
+            Object o = directory.getObject(tag);
+            if (o instanceof Date) {
+                return TikaDates.inYearBounds(((Date) o).toInstant()) ? (Date) 
o : null;
+            }
+            if (o == null) {
+                return null;
+            }
+            return TikaDates.parse(o.toString()).map(d -> 
Date.from(d.toInstant())).orElse(null);
         }
 
         private SimpleDateFormat getUnspecifiedTzDateFormat() {
@@ -734,7 +745,7 @@ public class ImageMetadataExtractor {
             // Date/Time Original overrides value from 
ExifDirectory.TAG_DATETIME
             Date original = null;
             if 
(directory.containsTag(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)) {
-                original = 
inBounds(directory.getDate(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL, GMT));
+                original = exifDate(directory, 
ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL);
                 // Unless we have GPS time we don't know the time zone so date 
must be set
                 // as ISO 8601 datetime without timezone suffix (no Z or +/-)
                 if (original != null) {
@@ -744,7 +755,7 @@ public class ImageMetadataExtractor {
                 }
             }
             if (directory.containsTag(ExifIFD0Directory.TAG_DATETIME)) {
-                Date datetime = 
inBounds(directory.getDate(ExifIFD0Directory.TAG_DATETIME, GMT));
+                Date datetime = exifDate(directory, 
ExifIFD0Directory.TAG_DATETIME);
                 if (datetime != null) {
                     String datetimeNoTimeZone = 
dateUnspecifiedTz.format(datetime);
                     metadata.set(TikaCoreProperties.MODIFIED, 
datetimeNoTimeZone);
@@ -807,6 +818,25 @@ public class ImageMetadataExtractor {
      * Maps EXIF Geo Tags onto the Tika Geo metadata namespace.
      */
     static class GeotagHandler implements DirectoryHandler {
+        // GpsDirectory.getGpsDate parses with the default locale's calendar
+        static Date gpsDate(GpsDirectory directory) {
+            String stamp = directory.getString(GpsDirectory.TAG_DATE_STAMP);
+            Rational[] time = 
directory.getRationalArray(GpsDirectory.TAG_TIME_STAMP);
+            if (stamp == null || time == null || time.length != 3) {
+                return null;
+            }
+            try {
+                LocalDate day = TikaDates.parse(stamp).map(d -> 
d.getLocalDateTime().toLocalDate()).orElse(null);
+                if (day == null) {
+                    return null;
+                }
+                LocalTime t = LocalTime.of(time[0].intValue(), 
time[1].intValue(), (int) time[2].doubleValue());
+                return Date.from(day.atTime(t).toInstant(ZoneOffset.UTC));
+            } catch (DateTimeException e) {
+                return null;
+            }
+        }
+
         public boolean supports(Class<? extends Directory> directoryType) {
             return directoryType == GpsDirectory.class;
         }
@@ -821,7 +851,7 @@ public class ImageMetadataExtractor {
                 metadata.set(TikaCoreProperties.LONGITUDE,
                         geoDecimalFormat.format(geoLocation.getLongitude()));
             }
-            Date gpsDate = ((GpsDirectory)directory).getGpsDate();
+            Date gpsDate = gpsDate((GpsDirectory) directory);
             if (gpsDate != null) {
                 metadata.set(Geographic.TIMESTAMP, gpsDate);
             }
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorLocaleTest.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorLocaleTest.java
index a4e015cb25..7bb9b1b246 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorLocaleTest.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorLocaleTest.java
@@ -18,10 +18,19 @@ package org.apache.tika.parser.image;
 
 import static org.junit.jupiter.api.Assertions.assertEquals;
 
+import java.io.InputStream;
 import java.util.Locale;
 
 import org.junit.jupiter.api.Test;
 import org.junit.jupiter.api.parallel.Isolated;
+import org.xml.sax.helpers.DefaultHandler;
+
+import org.apache.tika.io.TikaInputStream;
+import org.apache.tika.metadata.Geographic;
+import org.apache.tika.metadata.Metadata;
+import org.apache.tika.metadata.TIFF;
+import org.apache.tika.metadata.TikaCoreProperties;
+import org.apache.tika.parser.ParseContext;
 
 // sets the JVM-wide default locale, so it must not overlap other test classes
 @Isolated
@@ -38,4 +47,31 @@ public class ImageMetadataExtractorLocaleTest {
             Locale.setDefault(defaultLocale);
         }
     }
+
+    // the Japanese imperial calendar makes a default-locale SimpleDateFormat 
read 2009 as Reiwa 2009
+    @Test
+    public void testExifDatesIgnoreDefaultCalendar() throws Exception {
+        Locale defaultLocale = Locale.getDefault();
+        try {
+            Locale.setDefault(Locale.forLanguageTag("ja-JP-u-ca-japanese"));
+            Metadata metadata = parse("/test-documents/testJPEG_EXIF.jpg");
+            assertEquals("2009-08-11T09:09:45", 
metadata.get(TikaCoreProperties.CREATED));
+            assertEquals("2009-10-02T23:02:49", 
metadata.get(TikaCoreProperties.MODIFIED));
+            assertEquals("2009-08-11T09:09:45", 
metadata.get(TIFF.ORIGINAL_DATE));
+
+            metadata = parse("/test-documents/testJPEG_GEO_2.jpg");
+            assertEquals("2012-02-20T16:44:22Z", 
metadata.get(Geographic.TIMESTAMP));
+        } finally {
+            Locale.setDefault(defaultLocale);
+        }
+    }
+
+    private static Metadata parse(String resource) throws Exception {
+        Metadata metadata = new Metadata();
+        try (InputStream is = 
ImageMetadataExtractorLocaleTest.class.getResourceAsStream(resource);
+                TikaInputStream tis = TikaInputStream.get(is)) {
+            new JpegParser().parse(tis, new DefaultHandler(), metadata, new 
ParseContext());
+        }
+        return metadata;
+    }
 }
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorTest.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorTest.java
index ab93eab488..37b94732a6 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorTest.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorTest.java
@@ -25,11 +25,8 @@ import java.nio.ByteBuffer;
 import java.nio.charset.StandardCharsets;
 import java.util.ArrayList;
 import java.util.Arrays;
-import java.util.GregorianCalendar;
 import java.util.Iterator;
 import java.util.List;
-import java.util.Locale;
-import java.util.TimeZone;
 
 import com.drew.lang.Rational;
 import com.drew.metadata.Directory;
@@ -80,11 +77,7 @@ public class ImageMetadataExtractorTest {
     public void testExifHandlerParseDate() throws MetadataException {
         ExifSubIFDDirectory exif = Mockito.mock(ExifSubIFDDirectory.class);
         
Mockito.when(exif.containsTag(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn(true);
-        GregorianCalendar calendar = new 
GregorianCalendar(TimeZone.getTimeZone("UTC"), Locale.ROOT);
-        calendar.setTimeInMillis(0);
-        calendar.set(2000, 0, 1, 0, 0, 0);
-        Mockito.when(exif.getDate(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL, 
TimeZone.getTimeZone("GMT")))
-                .thenReturn(calendar.getTime());
+        
Mockito.when(exif.getObject(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn("2000:01:01
 00:00:00");
         Metadata metadata = new Metadata();
 
         new ImageMetadataExtractor.ExifHandler().handle(exif, metadata);
@@ -118,11 +111,7 @@ public class ImageMetadataExtractorTest {
     public void testExifHandlerParseDateFallback() throws MetadataException {
         ExifIFD0Directory exif = Mockito.mock(ExifIFD0Directory.class);
         
Mockito.when(exif.containsTag(ExifIFD0Directory.TAG_DATETIME)).thenReturn(true);
-        GregorianCalendar calendar = new 
GregorianCalendar(TimeZone.getTimeZone("UTC"), Locale.ROOT);
-        calendar.setTimeInMillis(0);
-        calendar.set(1999, 0, 1, 0, 0, 0);
-        Mockito.when(exif.getDate(ExifIFD0Directory.TAG_DATETIME, 
TimeZone.getTimeZone("GMT")))
-                .thenReturn(calendar.getTime());
+        
Mockito.when(exif.getObject(ExifIFD0Directory.TAG_DATETIME)).thenReturn("1999:01:01
 00:00:00");
         Metadata metadata = new Metadata();
 
         new ImageMetadataExtractor.ExifHandler().handle(exif, metadata);
@@ -134,11 +123,7 @@ public class ImageMetadataExtractorTest {
     public void testExifHandlerParseDateOutOfBounds() throws MetadataException 
{
         ExifSubIFDDirectory exif = Mockito.mock(ExifSubIFDDirectory.class);
         
Mockito.when(exif.containsTag(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn(true);
-        GregorianCalendar calendar = new 
GregorianCalendar(TimeZone.getTimeZone("UTC"), Locale.ROOT);
-        calendar.setTimeInMillis(0);
-        calendar.set(4, 11, 31, 23, 0, 0);
-        Mockito.when(exif.getDate(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL, 
TimeZone.getTimeZone("GMT")))
-                .thenReturn(calendar.getTime());
+        
Mockito.when(exif.getObject(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn("0004:12:31
 23:00:00");
         Metadata metadata = new Metadata();
         new ImageMetadataExtractor.ExifHandler().handle(exif, metadata);
         assertNull(metadata.get(TikaCoreProperties.CREATED));
@@ -149,7 +134,7 @@ public class ImageMetadataExtractorTest {
     public void testExifHandlerParseDateError() throws MetadataException {
         ExifIFD0Directory exif = Mockito.mock(ExifIFD0Directory.class);
         
Mockito.when(exif.containsTag(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn(true);
-        Mockito.when(exif.getDate(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL, 
TimeZone.getTimeZone("GMT"))).thenReturn(null);
+        
Mockito.when(exif.getObject(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn("
    :  :     :  :  ");
         Metadata metadata = new Metadata();
 
         new ImageMetadataExtractor.ExifHandler().handle(exif, metadata);
diff --git 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
index f2d9240fb6..a5b9c05992 100644
--- 
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
+++ 
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
@@ -80,8 +80,8 @@ public class JackcessParserTest extends TikaTest {
         }
     }
 
-    // TODO TIKA-4924: jackcess-encrypt uppercases cipher params in the 
default locale
-    @DisabledIfSystemProperty(named = "user.language", matches = "tr")
+    // TODO TIKA-4924: jackcess-encrypt uppercases cipher params in the 
default locale (tr/az dotless i)
+    @DisabledIfSystemProperty(named = "tika.test.locale", matches = 
"(tr|az)(-.*)?")
     @Test
     public void testPassword() throws Exception {
         ParseContext c = new ParseContext();
diff --git 
a/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/PerClientServerManager.java
 
b/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/PerClientServerManager.java
index 98a9b870f0..59390013f4 100644
--- 
a/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/PerClientServerManager.java
+++ 
b/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/PerClientServerManager.java
@@ -28,6 +28,7 @@ import java.nio.file.Files;
 import java.nio.file.Path;
 import java.util.ArrayList;
 import java.util.List;
+import java.util.Locale;
 import java.util.concurrent.TimeUnit;
 import java.util.concurrent.TimeoutException;
 
@@ -670,6 +671,7 @@ public class PerClientServerManager implements 
ServerManager {
         boolean hasLog4j = false;
         boolean hasActiveProcessorCount = false;
         boolean hasErrorFile = false;
+        boolean hasLocale = false;
         String origGCString = null;
         String newGCLogString = null;
 
@@ -692,6 +694,9 @@ public class PerClientServerManager implements 
ServerManager {
             if (arg.startsWith("-XX:ErrorFile=")) {
                 hasErrorFile = true;
             }
+            if (arg.startsWith("-Duser.language")) {
+                hasLocale = true;
+            }
             if (arg.startsWith("-Xloggc:")) {
                 origGCString = arg;
                 newGCLogString = arg.replace("${pipesClientId}", "id-" + 
clientId);
@@ -774,6 +779,10 @@ public class PerClientServerManager implements 
ServerManager {
             
commandLine.add("-Dlog4j.configurationFile=classpath:pipes-fork-server-default-log4j2.xml");
         }
         commandLine.add("-DpipesClientId=" + clientId);
+        // the fork parses like the parent would; a fresh JVM would take the 
OS locale instead
+        if (!hasLocale) {
+            
commandLine.addAll(ProcessUtils.defaultLocaleJvmArgs(Locale.getDefault()));
+        }
         commandLine.addAll(configArgs);
         commandLine.add("-Djava.io.tmpdir=" + tmpDir.toAbsolutePath());
         commandLine.add("org.apache.tika.pipes.core.server.PipesServer");
diff --git 
a/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/SharedServerManager.java
 
b/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/SharedServerManager.java
index 8374d837b4..f39383cb69 100644
--- 
a/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/SharedServerManager.java
+++ 
b/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/SharedServerManager.java
@@ -28,6 +28,7 @@ import java.nio.file.Files;
 import java.nio.file.Path;
 import java.util.ArrayList;
 import java.util.List;
+import java.util.Locale;
 import java.util.concurrent.TimeUnit;
 import java.util.concurrent.TimeoutException;
 import java.util.concurrent.atomic.AtomicLong;
@@ -504,6 +505,7 @@ public class SharedServerManager implements ServerManager {
         boolean hasExitOnOOM = false;
         boolean hasLog4j = false;
         boolean hasErrorFile = false;
+        boolean hasLocale = false;
 
         for (String arg : configArgs) {
             if (arg.startsWith("-Djava.awt.headless")) {
@@ -521,6 +523,9 @@ public class SharedServerManager implements ServerManager {
             if (arg.startsWith("-XX:ErrorFile=")) {
                 hasErrorFile = true;
             }
+            if (arg.startsWith("-Duser.language")) {
+                hasLocale = true;
+            }
         }
 
         // Direct native-crash dumps (hs_err_pid<N>.log) into tmpDir so
@@ -553,6 +558,10 @@ public class SharedServerManager implements ServerManager {
             
commandLine.add("-Dlog4j.configurationFile=classpath:pipes-fork-server-default-log4j2.xml");
         }
         commandLine.add("-DpipesClientId=shared");
+        // the fork parses like the parent would; a fresh JVM would take the 
OS locale instead
+        if (!hasLocale) {
+            
commandLine.addAll(ProcessUtils.defaultLocaleJvmArgs(Locale.getDefault()));
+        }
         commandLine.addAll(configArgs);
         commandLine.add("-Djava.io.tmpdir=" + tmpDir.toAbsolutePath());
         commandLine.add("org.apache.tika.pipes.core.server.PipesServer");
diff --git 
a/tika-pipes/tika-pipes-core/src/test/java/org/apache/tika/pipes/core/PerClientServerManagerLocaleTest.java
 
b/tika-pipes/tika-pipes-core/src/test/java/org/apache/tika/pipes/core/PerClientServerManagerLocaleTest.java
new file mode 100644
index 0000000000..57c99a3205
--- /dev/null
+++ 
b/tika-pipes/tika-pipes-core/src/test/java/org/apache/tika/pipes/core/PerClientServerManagerLocaleTest.java
@@ -0,0 +1,70 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.pipes.core;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+
+import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.Arrays;
+import java.util.List;
+import java.util.Locale;
+
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.io.TempDir;
+import org.junit.jupiter.api.parallel.Isolated;
+
+/** The fork gets the parent's default locale unless forkedJvmArgs set one. */
+// sets the JVM-wide default locale
+@Isolated
+public class PerClientServerManagerLocaleTest {
+
+    @TempDir
+    Path tmp;
+
+    private List<String> userArgs(String... forkedJvmArgs) throws Exception {
+        PipesConfig pipesConfig = new PipesConfig();
+        pipesConfig.setForkedJvmArgs(new 
ArrayList<>(Arrays.asList(forkedJvmArgs)));
+        PerClientServerManager manager =
+                new PerClientServerManager(pipesConfig, new byte[0], 0);
+        return Arrays.stream(manager.getCommandline(tmp))
+                .filter(a -> a.startsWith("-Duser.")).toList();
+    }
+
+    @Test
+    public void testParentLocalePropagates() throws Exception {
+        Locale defaultLocale = Locale.getDefault();
+        try {
+            
Locale.setDefault(Locale.forLanguageTag("th-TH-u-nu-thai-x-lvariant-TH"));
+            assertEquals(List.of("-Duser.language=th", "-Duser.country=TH", 
"-Duser.variant=TH",
+                    "-Duser.extensions=u-nu-thai"), userArgs());
+            Locale.setDefault(Locale.forLanguageTag("sr-Latn-RS"));
+            assertEquals(List.of("-Duser.language=sr", "-Duser.script=Latn", 
"-Duser.country=RS"),
+                    userArgs());
+            Locale.setDefault(Locale.ROOT);
+            assertEquals(List.of(), userArgs());
+        } finally {
+            Locale.setDefault(defaultLocale);
+        }
+    }
+
+    @Test
+    public void testConfiguredLocaleWins() throws Exception {
+        assertEquals(List.of("-Duser.language=tr", "-Duser.country=TR"),
+                userArgs("-Duser.language=tr", "-Duser.country=TR"));
+    }
+}
diff --git a/tika-test-support/pom.xml b/tika-test-support/pom.xml
new file mode 100644
index 0000000000..88946b628f
--- /dev/null
+++ b/tika-test-support/pom.xml
@@ -0,0 +1,56 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<!--
+  Licensed to the Apache Software Foundation (ASF) under one
+  or more contributor license agreements.  See the NOTICE file
+  distributed with this work for additional information
+  regarding copyright ownership.  The ASF licenses this file
+  to you under the Apache License, Version 2.0 (the
+  "License"); you may not use this file except in compliance
+  with the License.  You may obtain a copy of the License at
+
+    http://www.apache.org/licenses/LICENSE-2.0
+
+  Unless required by applicable law or agreed to in writing,
+  software distributed under the License is distributed on an
+  "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+  KIND, either express or implied.  See the License for the
+  specific language governing permissions and limitations
+  under the License.
+-->
+<project xmlns="http://maven.apache.org/POM/4.0.0"; 
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"; 
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 
https://maven.apache.org/xsd/maven-4.0.0.xsd";>
+  <modelVersion>4.0.0</modelVersion>
+  <parent>
+    <groupId>org.apache.tika</groupId>
+    <artifactId>tika-parent</artifactId>
+    <version>${revision}</version>
+    <relativePath>../tika-parent/pom.xml</relativePath>
+  </parent>
+  <artifactId>tika-test-support</artifactId>
+  <name>Apache Tika test support</name>
+  <description>
+    JUnit Platform listeners on every module's test classpath: a random default
+    locale per test JVM, reported on failure.
+  </description>
+  <url>https://tika.apache.org</url>
+  <dependencies>
+    <dependency>
+      <groupId>org.junit.platform</groupId>
+      <artifactId>junit-platform-launcher</artifactId>
+    </dependency>
+  </dependencies>
+  <build>
+    <plugins>
+      <plugin>
+        <groupId>org.apache.maven.plugins</groupId>
+        <artifactId>maven-jar-plugin</artifactId>
+        <configuration>
+          <archive>
+            <manifestEntries>
+              
<Automatic-Module-Name>org.apache.tika.test.support</Automatic-Module-Name>
+            </manifestEntries>
+          </archive>
+        </configuration>
+      </plugin>
+    </plugins>
+  </build>
+</project>
diff --git 
a/tika-test-support/src/main/java/org/apache/tika/test/RandomLocaleListener.java
 
b/tika-test-support/src/main/java/org/apache/tika/test/RandomLocaleListener.java
new file mode 100644
index 0000000000..b633064735
--- /dev/null
+++ 
b/tika-test-support/src/main/java/org/apache/tika/test/RandomLocaleListener.java
@@ -0,0 +1,122 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.test;
+
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
+import java.util.concurrent.ThreadLocalRandom;
+
+import org.junit.platform.engine.TestExecutionResult;
+import org.junit.platform.launcher.LauncherSession;
+import org.junit.platform.launcher.LauncherSessionListener;
+import org.junit.platform.launcher.TestExecutionListener;
+import org.junit.platform.launcher.TestIdentifier;
+
+/**
+ * With {@code -Dtika.test.locale=random}, runs the test JVM under a random 
default locale, as
+ * Lucene's test framework does, so locale-sensitive code paths are exercised 
across the
+ * JDK's whole locale set over time rather than in one pinned locale. CI 
passes that flag;
+ * a plain build keeps the JVM's own locale, so release and developer builds 
are
+ * deterministic. Registered through {@code META-INF/services}; the launcher 
picks it up
+ * from the test classpath.
+ * <p>
+ * {@code -Dtika.test.locale=<language tag>} pins the locale (as printed by a 
failed run).
+ * The effective language tag is published under the same property, and every 
failure
+ * carries it as a suppressed exception so a surefire report shows how to 
reproduce.
+ */
+public class RandomLocaleListener implements LauncherSessionListener, 
TestExecutionListener {
+
+    public static final String PROPERTY = "tika.test.locale";
+
+    /**
+     * Half of the random picks come from here: locales whose casing, digit, 
calendar or
+     * shaping rules differ from en-US in ways code often overlooks, and which 
uniform
+     * sampling would each reach only once in several hundred runs.
+     */
+    static final List<Locale> PRIORITY_LOCALES = List.of(
+            Locale.forLanguageTag("tr-TR"),
+            Locale.forLanguageTag("az-Latn-AZ"),
+            Locale.forLanguageTag("th-TH-u-nu-thai-x-lvariant-TH"),
+            Locale.forLanguageTag("ja-JP-u-ca-japanese-x-lvariant-JP"),
+            Locale.forLanguageTag("ar-EG"),
+            Locale.forLanguageTag("de-DE"));
+
+    private static volatile boolean reported;
+
+    @Override
+    public void launcherSessionOpened(LauncherSession session) {
+        Locale locale = choose(System.getProperty(PROPERTY));
+        Locale.setDefault(locale);
+        System.setProperty(PROPERTY, locale.toLanguageTag());
+        System.out.println("[tika] default locale " + locale.toLanguageTag() +
+                " (reproduce with -D" + PROPERTY + "=" + 
locale.toLanguageTag() + ")");
+    }
+
+    static Locale choose(String setting) {
+        if (setting == null || setting.isEmpty() || "system".equals(setting)) {
+            return Locale.getDefault();
+        }
+        if ("random".equals(setting)) {
+            ThreadLocalRandom random = ThreadLocalRandom.current();
+            List<Locale> candidates = random.nextBoolean() ? PRIORITY_LOCALES 
: candidates();
+            return candidates.get(random.nextInt(candidates.size()));
+        }
+        Locale locale = Locale.forLanguageTag(setting);
+        if (locale.getLanguage().isEmpty()) {
+            throw new IllegalArgumentException("not a language tag: -D" + 
PROPERTY + "=" + setting);
+        }
+        return locale;
+    }
+
+    /** Available locales whose language tag reproduces them; no_NO_NY, for 
one, does not. */
+    static List<Locale> candidates() {
+        List<Locale> candidates = new ArrayList<>();
+        for (Locale locale : Locale.getAvailableLocales()) {
+            if (!locale.getLanguage().isEmpty() &&
+                    
locale.equals(Locale.forLanguageTag(locale.toLanguageTag()))) {
+                candidates.add(locale);
+            }
+        }
+        return candidates;
+    }
+
+    @Override
+    public void executionFinished(TestIdentifier identifier, 
TestExecutionResult result) {
+        if (result.getStatus() != TestExecutionResult.Status.FAILED) {
+            return;
+        }
+        String tag = System.getProperty(PROPERTY);
+        if (tag == null) {
+            return;
+        }
+        result.getThrowable().ifPresent(t -> t.addSuppressed(new 
LocaleNote(tag)));
+        if (!reported) {
+            reported = true;
+            System.err.println("[tika] failure under default locale " + tag +
+                    "; reproduce with -D" + PROPERTY + "=" + tag);
+        }
+    }
+
+    /** Attached to a failure so the reproduction hint survives into the 
surefire report. */
+    public static final class LocaleNote extends Exception {
+        LocaleNote(String tag) {
+            super("default locale was " + tag + "; reproduce with -D" + 
PROPERTY + "=" + tag,
+                    null, false, false);
+        }
+    }
+}
diff --git 
a/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.LauncherSessionListener
 
b/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.LauncherSessionListener
new file mode 100644
index 0000000000..f4c1a97825
--- /dev/null
+++ 
b/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.LauncherSessionListener
@@ -0,0 +1,17 @@
+#
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements.  See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License.  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+org.apache.tika.test.RandomLocaleListener
diff --git 
a/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.TestExecutionListener
 
b/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.TestExecutionListener
new file mode 100644
index 0000000000..f4c1a97825
--- /dev/null
+++ 
b/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.TestExecutionListener
@@ -0,0 +1,17 @@
+#
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements.  See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License.  You may obtain a copy of the License at
+#
+#      http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+org.apache.tika.test.RandomLocaleListener

Reply via email to