This is an automated email from the ASF dual-hosted git repository.
tballison pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/tika.git
The following commit(s) were added to refs/heads/main by this push:
new 3db3c2e87f TIKA-4920: run tests under a random default locale; forks
inherit the… (#3247)
3db3c2e87f is described below
commit 3db3c2e87f8bd19362145611b5043872663938a7
Author: Tim Allison <[email protected]>
AuthorDate: Tue Oct 6 17:34:52 2026 -0400
TIKA-4920: run tests under a random default locale; forks inherit the…
(#3247)
---
.github/workflows/main-jdk17-build.yml | 3 +-
.github/workflows/main-jdk17-locale-build.yml | 78 -------------
.github/workflows/main-jdk17-windows-build.yml | 20 ++--
.github/workflows/main-jdk21-build.yml | 2 +-
.github/workflows/main-jdk25-build.yml | 2 +-
.skills/devs/development/SKILL.md | 6 +
CHANGES.txt | 8 ++
pom.xml | 1 +
.../java/org/apache/tika/utils/ProcessUtils.java | 36 ++++++
.../org/apache/tika/DefaultLocaleCanaryTest.java | 37 -------
.../apache/tika/test/RandomLocaleListenerTest.java | 64 +++++++++++
tika-ml/tika-ml-chardetect/pom.xml | 7 --
.../chardetect/tools/BuildCharsetTrainingData.java | 35 +++---
.../chardetect/tools/DiagnoseDiscrimination.java | 2 +-
tika-ml/tika-ml-junkdetect-tools/pom.xml | 7 --
.../ml/junkdetect/tools/BoundaryBigramAudit.java | 4 +-
.../ml/junkdetect/tools/BuildJunkTrainingData.java | 40 +++----
.../tika/ml/junkdetect/tools/DebugScriptRuns.java | 7 +-
.../ml/junkdetect/tools/LineScriptFractions.java | 8 +-
.../tika/ml/junkdetect/tools/ScriptCensus.java | 9 +-
.../tika/ml/junkdetect/tools/TrainJunkModel.java | 34 +++---
.../tools/BuildJunkAugmentationData.java | 34 +++---
tika-ml/tika-ml-junkdetect/pom.xml | 7 --
.../ml/junkdetect/LatinSiblingComparisonTest.java | 9 +-
tika-parent/checkstyle.xml | 4 +-
tika-parent/pom.xml | 27 ++++-
.../apache/tika/parser/grib/GribParserTest.java | 8 ++
.../tika/parser/netcdf/NetCDFParserTest.java | 8 ++
.../tika/parser/ocr/TesseractOCRParserTest.java | 4 +-
.../org/apache/tika/parser/sas/SAS7BDATParser.java | 4 +-
.../apache/tika/parser/sas/SAS7BDATParserTest.java | 8 +-
.../tika/parser/image/ImageMetadataExtractor.java | 44 ++++++--
.../image/ImageMetadataExtractorLocaleTest.java | 36 ++++++
.../parser/image/ImageMetadataExtractorTest.java | 23 +---
.../tika/parser/microsoft/JackcessParserTest.java | 4 +-
.../tika/pipes/core/PerClientServerManager.java | 9 ++
.../tika/pipes/core/SharedServerManager.java | 9 ++
.../core/PerClientServerManagerLocaleTest.java | 70 ++++++++++++
tika-test-support/pom.xml | 56 ++++++++++
.../org/apache/tika/test/RandomLocaleListener.java | 122 +++++++++++++++++++++
...junit.platform.launcher.LauncherSessionListener | 17 +++
...g.junit.platform.launcher.TestExecutionListener | 17 +++
42 files changed, 651 insertions(+), 279 deletions(-)
diff --git a/.github/workflows/main-jdk17-build.yml
b/.github/workflows/main-jdk17-build.yml
index ffb7f16445..033a5b6888 100644
--- a/.github/workflows/main-jdk17-build.yml
+++ b/.github/workflows/main-jdk17-build.yml
@@ -100,6 +100,7 @@ jobs:
run: |
mvn install -Pfast -pl :tika-annotation-processor -am -B -q
mvn clean test install ${{ github.event_name == 'push' &&
'javadoc:aggregate' || '' }} -Pci -T1C \
+ -Dtika.test.locale=random \
-pl "$(echo "$IT_MODULES" | sed 's/:/!:/g')" \
-B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
@@ -140,7 +141,7 @@ jobs:
# so without this they would drop out of license checking entirely.
- name: Run integration tests
run: |
- mvn clean apache-rat:check test -Pci \
+ mvn clean apache-rat:check test -Pci -Dtika.test.locale=random \
-pl "$IT_MODULES" \
-B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git a/.github/workflows/main-jdk17-locale-build.yml
b/.github/workflows/main-jdk17-locale-build.yml
deleted file mode 100644
index 899bcd5687..0000000000
--- a/.github/workflows/main-jdk17-locale-build.yml
+++ /dev/null
@@ -1,78 +0,0 @@
-#
-# Licensed to the Apache Software Foundation (ASF) under one or more
-# contributor license agreements. See the NOTICE file distributed with
-# this work for additional information regarding copyright ownership.
-# The ASF licenses this file to You under the Apache License, Version 2.0
-# (the "License"); you may not use this file except in compliance with
-# the License. You may obtain a copy of the License at
-#
-# http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-#
-
-# tr_TR is the highest-value locale to test in Java: dotless-i means
-# "TIFF".toLowerCase() is not "tiff" unless the call passes Locale.ROOT.
-#
-# Linux, not Windows: locale bugs are JVM-level, so the 2x-cost Windows runner
-# buys nothing here -- main-jdk17-windows-build covers the OS-specific surface.
-#
-# Locale via JAVA_TOOL_OPTIONS, which every JVM (surefire forks included)
reads at
-# startup. Not LANG/LC_ALL: the JVM falls back to en_US when the locale is not
-# generated on the runner. Not -Duser.language on the mvn command line:
surefire
-# sets that in the fork only after the default Locale is fixed, so it tests
nothing.
-# DefaultLocaleCanaryTest fails the build if the locale does not reach the
fork.
-#
-# push-only, like the jdk21/jdk25 builds: locale regressions are rare and not
-# usually PR-specific, so a full reactor build per PR is not worth the cost.
-name: main jdk17 locale build (tr_TR)
-
-on:
- push:
- branches: [ main ]
- paths-ignore:
- - 'docs/**'
-
-jobs:
- build:
- runs-on: ubuntu-latest
- timeout-minutes: 60
- strategy:
- matrix:
- java: [ '17' ]
-
- steps:
- - uses: actions/checkout@v7
- - name: Set up JDK ${{ matrix.java }}
- uses: actions/setup-java@v6
- with:
- distribution: 'temurin'
- java-version: ${{ matrix.java }}
- - name: Cache Maven repository
- # Explicit, not setup-java's cache: 'maven', so our own snapshots stay
out of the tarball.
- uses: actions/cache@v6
- with:
- path: |
- ~/.m2/repository
- !~/.m2/repository/org/apache/tika
- key: maven-${{ runner.os }}-${{ hashFiles('**/pom.xml') }}
- restore-keys: maven-${{ runner.os }}-
- - name: Install external tools
- run: sudo apt-get update && sudo apt-get install -y ffmpeg
libimage-exiftool-perl
- # The Docker-backed integration tests spin up
Elasticsearch/OpenSearch/Solr/Kafka/RustFS
- # for ~7.5 min and carry no locale signal, so they are excluded here;
the main jdk17
- # build runs them. If a new testcontainers module appears, add it to
this list --
- # forgetting only makes this job slower, it does not weaken it.
- - name: Build with Maven (tr_TR locale)
- env:
- JAVA_TOOL_OPTIONS: -Duser.language=tr -Duser.country=TR
- TIKA_EXPECTED_LOCALE: tr-TR
- run: |
- mvn install -Pfast -pl :tika-annotation-processor -am -B -q
- mvn clean test install -Pci -T1C \
- -pl
'!:tika-pipes-es-integration-tests,!:tika-pipes-kafka-integration-tests,!:tika-pipes-opensearch-integration-tests,!:tika-pipes-s3-integration-tests,!:tika-pipes-solr-integration-tests'
\
- -B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git a/.github/workflows/main-jdk17-windows-build.yml
b/.github/workflows/main-jdk17-windows-build.yml
index 4613496528..ac8de6c2b5 100644
--- a/.github/workflows/main-jdk17-windows-build.yml
+++ b/.github/workflows/main-jdk17-windows-build.yml
@@ -17,20 +17,21 @@
# The one Windows job. Two things it uniquely covers:
# - path handling: the checkout dir below deliberately contains a space
-# - an alternate (non-en_US) locale
+# - child JVMs (tika-server, pipes forks) started under a non-en_US locale
#
-# Locale is set with JAVA_TOOL_OPTIONS, which every JVM (surefire forks
included)
-# reads at startup. NOT LANG/LC_ALL: the Windows JVM reads the OS locale via
Win32
-# and ignores them. NOT -Duser.language on the mvn command line: surefire sets
that
-# in the fork only after the default Locale is fixed, so the tests ran in
en_US.
-# DefaultLocaleCanaryTest fails the build if the locale does not reach the
fork.
+# The Linux jobs run in-process tests under a random default locale
+# (-Dtika.test.locale=random, RandomLocaleListener); this one runs them under
+# de_DE. Pipes forks receive their parent's locale as -D flags, but JVMs the
+# tests start themselves (tika-server) inherit only the environment, hence
+# JAVA_TOOL_OPTIONS here, which every JVM reads at startup.
+# NOT LANG/LC_ALL: the Windows JVM reads the OS locale via Win32 and ignores
them.
#
-# push-only, like the jdk21/jdk25 and tr_TR builds. This is the most expensive
+# push-only, like the jdk21/jdk25 builds. This is the most expensive
# job in CI -- a full reactor on a 2x-cost runner, ~48 min -- and it gated PR
# wall clock all by itself while every Linux job finished in ~22. Windows-only
# regressions are real but rare, so they are caught on main within the hour
# rather than paid for on every PR push.
-name: main jdk17 windows build (de_DE)
+name: main jdk17 windows build
on:
push:
@@ -65,11 +66,10 @@ jobs:
!~/.m2/repository/org/apache/tika
key: maven-${{ runner.os }}-${{ hashFiles('**/pom.xml') }}
restore-keys: maven-${{ runner.os }}-
- - name: Build with Maven (de_DE locale)
+ - name: Build with Maven (de_DE child JVMs)
working-directory: 'tika build dir'
env:
JAVA_TOOL_OPTIONS: -Duser.language=de -Duser.country=DE
- TIKA_EXPECTED_LOCALE: de-DE
# No -T1C here (TIKA-4867): this is the heaviest job (locale + e2e +
javadoc)
# on a 4-core runner; parallel modules starved the tika-server
integration
# tests into startup timeouts. Serial reactor also needs no annotation-
diff --git a/.github/workflows/main-jdk21-build.yml
b/.github/workflows/main-jdk21-build.yml
index dba342210d..3aa717983a 100644
--- a/.github/workflows/main-jdk21-build.yml
+++ b/.github/workflows/main-jdk21-build.yml
@@ -48,4 +48,4 @@ jobs:
key: maven-${{ runner.os }}-${{ hashFiles('**/pom.xml') }}
restore-keys: maven-${{ runner.os }}-
- name: Build with Maven
- run: mvn install -Pfast -pl :tika-annotation-processor -am -B -q &&
mvn clean test install javadoc:aggregate -Pci -T1C -B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
+ run: mvn install -Pfast -pl :tika-annotation-processor -am -B -q &&
mvn clean test install javadoc:aggregate -Pci -T1C -Dtika.test.locale=random -B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git a/.github/workflows/main-jdk25-build.yml
b/.github/workflows/main-jdk25-build.yml
index 690304a24d..58187e295b 100644
--- a/.github/workflows/main-jdk25-build.yml
+++ b/.github/workflows/main-jdk25-build.yml
@@ -48,4 +48,4 @@ jobs:
key: maven-${{ runner.os }}-${{ hashFiles('**/pom.xml') }}
restore-keys: maven-${{ runner.os }}-
- name: Build with Maven
- run: mvn install -Pfast -pl :tika-annotation-processor -am -B -q &&
mvn clean test install javadoc:aggregate -Pci -T1C -B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
+ run: mvn install -Pfast -pl :tika-annotation-processor -am -B -q &&
mvn clean test install javadoc:aggregate -Pci -T1C -Dtika.test.locale=random -B
"-Dorg.slf4j.simpleLogger.log.org.apache.maven.cli.transfer.Slf4jMavenTransferListener=warn"
diff --git a/.skills/devs/development/SKILL.md
b/.skills/devs/development/SKILL.md
index 9a04a78968..863c05aeb9 100644
--- a/.skills/devs/development/SKILL.md
+++ b/.skills/devs/development/SKILL.md
@@ -211,6 +211,12 @@ them back. Anything hot enough to need a real seek gets a
channel from
CONCATENATE-only bugs).
- Keep tests non-duplicative: don't add a test whose failure another test
already guarantees.
+- CI runs every test JVM under a random default locale
(`-Dtika.test.locale=random`;
+ `RandomLocaleListener` in `tika-test-support`, wired through tika-parent). A
plain
+ build keeps the JVM's own locale. A CI failure's stack trace and the fork's
stderr
+ name the locale; reproduce with `-Dtika.test.locale=<tag>`. A test that
fails only
+ in some locales is a bug in the code under test (use `Locale.ROOT`), not a
+ reason to pin `Locale.US` in the test.
- Where there's bang for the buck, prefer parameterized tests over
copy-pasted cases, randomized inputs over hand-picked ones (log the seed
so failures reproduce), and fuzzing for parsers and format/boundary
diff --git a/CHANGES.txt b/CHANGES.txt
index ef89983799..5a2ece3ec1 100644
--- a/CHANGES.txt
+++ b/CHANGES.txt
@@ -240,6 +240,14 @@ Release 4.1.0 - 9/26/2026
* The Ignite config store quotes its table name, so lookups work under any
default locale (TIKA-4922).
+ * CI runs each test JVM under a random default locale
(-Dtika.test.locale=random,
+ tika-test-support's RandomLocaleListener); a failure names the locale and
+ -Dtika.test.locale=<tag> reproduces it. Replaces the fixed tr_TR CI job
(TIKA-4920).
+
+ * Pipes forks now start with the parent JVM's default locale (-Duser.*)
+ unless forkedJvmArgs set user.language; a fresh JVM took the OS locale,
+ so forked parsing could differ from in-process parsing (TIKA-4920).
+
* tika-annotation-processor is no longer a transitive dependency of
tika-pipes-core or the tika-langdetect modules (provided scope).
diff --git a/pom.xml b/pom.xml
index 52b72c875b..8d9db9bb8b 100644
--- a/pom.xml
+++ b/pom.xml
@@ -37,6 +37,7 @@
<modules>
<module>tika-parent</module>
<module>tika-bom</module>
+ <module>tika-test-support</module>
<module>tika-core</module>
<module>tika-annotation-processor</module>
<module>tika-serialization</module>
diff --git a/tika-core/src/main/java/org/apache/tika/utils/ProcessUtils.java
b/tika-core/src/main/java/org/apache/tika/utils/ProcessUtils.java
index e1916b43cc..244742cb74 100644
--- a/tika-core/src/main/java/org/apache/tika/utils/ProcessUtils.java
+++ b/tika-core/src/main/java/org/apache/tika/utils/ProcessUtils.java
@@ -20,6 +20,9 @@ package org.apache.tika.utils;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
import java.util.Optional;
import java.util.UUID;
import java.util.concurrent.ConcurrentHashMap;
@@ -84,6 +87,39 @@ public class ProcessUtils {
return arg;
}
+ /**
+ * The -D arguments that give a child JVM {@code locale} as its default: a
fresh JVM reads
+ * the OS locale, not its parent's, and a command-line -D beats
JAVA_TOOL_OPTIONS.
+ * Empty for {@link Locale#ROOT}.
+ */
+ public static List<String> defaultLocaleJvmArgs(Locale locale) {
+ List<String> args = new ArrayList<>();
+ if (locale.getLanguage().isEmpty()) {
+ return args;
+ }
+ args.add("-Duser.language=" + locale.getLanguage());
+ if (!locale.getScript().isEmpty()) {
+ args.add("-Duser.script=" + locale.getScript());
+ }
+ if (!locale.getCountry().isEmpty()) {
+ args.add("-Duser.country=" + locale.getCountry());
+ }
+ if (!locale.getVariant().isEmpty()) {
+ args.add("-Duser.variant=" + locale.getVariant());
+ }
+ StringBuilder extensions = new StringBuilder();
+ for (char key : locale.getExtensionKeys()) {
+ if (extensions.length() > 0) {
+ extensions.append('-');
+ }
+
extensions.append(key).append('-').append(locale.getExtension(key));
+ }
+ if (extensions.length() > 0) {
+ args.add("-Duser.extensions=" + extensions);
+ }
+ return args;
+ }
+
public static String unescapeCommandLine(String arg) {
if (arg.contains(" ") && SystemUtils.IS_OS_WINDOWS &&
(arg.startsWith("\"") && arg.endsWith("\""))) {
diff --git
a/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java
b/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java
deleted file mode 100644
index e02b5a2c1f..0000000000
--- a/tika-core/src/test/java/org/apache/tika/DefaultLocaleCanaryTest.java
+++ /dev/null
@@ -1,37 +0,0 @@
-/*
- * Licensed to the Apache Software Foundation (ASF) under one or more
- * contributor license agreements. See the NOTICE file distributed with
- * this work for additional information regarding copyright ownership.
- * The ASF licenses this file to You under the Apache License, Version 2.0
- * (the "License"); you may not use this file except in compliance with
- * the License. You may obtain a copy of the License at
- *
- * http://www.apache.org/licenses/LICENSE-2.0
- *
- * Unless required by applicable law or agreed to in writing, software
- * distributed under the License is distributed on an "AS IS" BASIS,
- * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
- * See the License for the specific language governing permissions and
- * limitations under the License.
- */
-package org.apache.tika;
-
-import static org.junit.jupiter.api.Assertions.assertEquals;
-
-import java.util.Locale;
-
-import org.junit.jupiter.api.Test;
-import org.junit.jupiter.api.condition.EnabledIfEnvironmentVariable;
-
-/**
- * Fails a locale CI job whose locale never reached the forked test JVM:
-Duser.language on
- * the mvn command line becomes a system property only after the default
Locale is fixed.
- */
-public class DefaultLocaleCanaryTest {
-
- @Test
- @EnabledIfEnvironmentVariable(named = "TIKA_EXPECTED_LOCALE", matches =
".+")
- public void testDefaultLocaleMatchesExpected() {
- assertEquals(System.getenv("TIKA_EXPECTED_LOCALE"),
Locale.getDefault().toLanguageTag());
- }
-}
diff --git
a/tika-core/src/test/java/org/apache/tika/test/RandomLocaleListenerTest.java
b/tika-core/src/test/java/org/apache/tika/test/RandomLocaleListenerTest.java
new file mode 100644
index 0000000000..eeec5d1982
--- /dev/null
+++ b/tika-core/src/test/java/org/apache/tika/test/RandomLocaleListenerTest.java
@@ -0,0 +1,64 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.test;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+import static org.junit.jupiter.api.Assertions.assertSame;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+import java.util.List;
+import java.util.Locale;
+
+import org.junit.jupiter.api.Test;
+
+/**
+ * Also the canary: the listener reaches every module through tika-parent, and
this test fails
+ * the build if it stops doing so.
+ */
+public class RandomLocaleListenerTest {
+
+ @Test
+ public void testListenerAppliedToThisJvm() {
+ String tag = System.getProperty(RandomLocaleListener.PROPERTY);
+ assertNotNull(tag, "listener did not run");
+ assertEquals(tag, Locale.getDefault().toLanguageTag());
+ }
+
+ @Test
+ public void testChoose() {
+ assertEquals(Locale.forLanguageTag("tr-TR"),
RandomLocaleListener.choose("tr-TR"));
+ assertSame(Locale.getDefault(), RandomLocaleListener.choose("system"));
+ assertSame(Locale.getDefault(), RandomLocaleListener.choose(null));
+ assertSame(Locale.getDefault(), RandomLocaleListener.choose(""));
+ assertThrows(IllegalArgumentException.class, () ->
RandomLocaleListener.choose("?"));
+ List<Locale> candidates = RandomLocaleListener.candidates();
+ assertTrue(candidates.size() > 500, "only " + candidates.size() + "
candidates");
+
assertTrue(candidates.containsAll(RandomLocaleListener.PRIORITY_LOCALES));
+ // the tag printed on failure must reproduce the pick, variants and
extensions included
+ for (Locale candidate : candidates) {
+ assertFalse(candidate.getLanguage().isEmpty());
+ assertEquals(candidate,
RandomLocaleListener.choose(candidate.toLanguageTag()),
+ candidate.toString());
+ }
+ for (int i = 0; i < 20; i++) {
+
assertTrue(candidates.contains(RandomLocaleListener.choose("random")));
+ }
+ }
+}
diff --git a/tika-ml/tika-ml-chardetect/pom.xml
b/tika-ml/tika-ml-chardetect/pom.xml
index 8ee7abe212..15f943f4a9 100644
--- a/tika-ml/tika-ml-chardetect/pom.xml
+++ b/tika-ml/tika-ml-chardetect/pom.xml
@@ -96,13 +96,6 @@
</configuration>
</plugin>
<!-- Tools use System.out/printf freely — suppress forbidden-apis for
this module -->
- <plugin>
- <groupId>de.thetaphi</groupId>
- <artifactId>forbiddenapis</artifactId>
- <configuration>
- <skip>true</skip>
- </configuration>
- </plugin>
</plugins>
</build>
diff --git
a/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/BuildCharsetTrainingData.java
b/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/BuildCharsetTrainingData.java
index f863456a69..fecb75fe5c 100644
---
a/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/BuildCharsetTrainingData.java
+++
b/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/BuildCharsetTrainingData.java
@@ -38,6 +38,7 @@ import java.util.HashMap;
import java.util.HashSet;
import java.util.LinkedHashMap;
import java.util.List;
+import java.util.Locale;
import java.util.Map;
import java.util.Random;
import java.util.Set;
@@ -513,19 +514,19 @@ public class BuildCharsetTrainingData {
System.out.println("=== BuildCharsetTrainingData ===");
System.out.println(" madlad-dir: " + madladDir);
System.out.println(" output-dir: " + outputDir);
- System.out.printf (" sample cap: %,d%n", sampleCap);
- System.out.printf (" byte budget: %,d%n", byteBudget);
- System.out.printf (" chunk: %d–%d bytes seed=%d%n",
minChunk, maxChunk, seed);
- System.out.printf (" unicode langs: %d (from %s)%n",
+ System.out.printf(Locale.ROOT, " sample cap: %,d%n",
sampleCap);
+ System.out.printf(Locale.ROOT, " byte budget: %,d%n",
byteBudget);
+ System.out.printf(Locale.ROOT, " chunk: %d–%d bytes
seed=%d%n", minChunk, maxChunk, seed);
+ System.out.printf(Locale.ROOT, " unicode langs: %d (from %s)%n",
unicodeLangs.size(), unicodeLangsFile.getFileName());
- System.out.printf (" charsets: %d%n%n",
targetCharsets.size());
+ System.out.printf(Locale.ROOT, " charsets: %d%n%n",
targetCharsets.size());
// Ambiguity gate: for each SBCS charset, precompute encoders for all
// other SBCS charsets. A chunk is dropped if any rival produces
// byte-for-byte identical output — such chunks carry no discriminative
// signal and actively confuse the model.
Map<String, List<CharsetEncoder>> sbcsRivals =
buildSbcsRivals(targetCharsets);
- System.out.printf(" ambiguity-gate: %d SBCS charsets compared
pairwise%n%n",
+ System.out.printf(Locale.ROOT, " ambiguity-gate: %d SBCS
charsets compared pairwise%n%n",
sbcsRivals.size());
// charset label → split name → sample count (for manifest)
@@ -540,7 +541,7 @@ public class BuildCharsetTrainingData {
String[] splits = {"train", "devtest", "test"};
int[] splitCaps = {sampleCap, sampleCap, sampleCap};
long[] splitBudgets = {byteBudget, byteBudget / 5, byteBudget /
5};
- System.out.printf("%s (%s)%s%n", label, javaName,
+ System.out.printf(Locale.ROOT, "%s (%s)%s%n", label, javaName,
structOnly ? " [structural-only: skipping train]" : "");
// Determine contributing languages.
@@ -568,17 +569,17 @@ public class BuildCharsetTrainingData {
? Math.max(5_000, UNICODE_SENTENCE_BUDGET / nLangs)
: Math.min(MAX_LOAD_CAP_PER_LANG, LEGACY_SENTENCE_BUDGET /
nLangs);
List<String> allSentences = new ArrayList<>();
- System.out.printf(" Contributing languages (%d),
perLangCap=%,d:%n",
+ System.out.printf(Locale.ROOT, " Contributing languages (%d),
perLangCap=%,d:%n",
nLangs, perLangCap);
for (String lang : langs) {
Path langDir = madladDir.resolve(lang);
List<String> sents = loadMadladSentences(langDir, perLangCap);
if (sents.isEmpty()) {
- System.out.printf(" %-6s: ** 0 sentences — MISSING DATA
**%n", lang);
+ System.out.printf(Locale.ROOT, " %-6s: ** 0 sentences —
MISSING DATA **%n", lang);
} else {
long totalChars = 0;
for (String s : sents) totalChars += s.length();
- System.out.printf(" %-6s: %,8d sentences avg_len=%,.0f
chars%n",
+ System.out.printf(Locale.ROOT, " %-6s: %,8d sentences
avg_len=%,.0f chars%n",
lang, sents.size(), (double) totalChars /
sents.size());
}
allSentences.addAll(sents);
@@ -632,10 +633,10 @@ public class BuildCharsetTrainingData {
splitBytes.put(split, totalBytes);
double budgetPct = 100.0 * totalBytes / budget;
if (ambiguousDropped > 0) {
- System.out.printf(" %s: %,d samples %,d bytes (%.1f%%
of budget) (%,d ambiguous-dropped)%n",
+ System.out.printf(Locale.ROOT, " %s: %,d samples %,d
bytes (%.1f%% of budget) (%,d ambiguous-dropped)%n",
split, written, totalBytes, budgetPct,
ambiguousDropped);
} else {
- System.out.printf(" %s: %,d samples %,d bytes (%.1f%%
of budget)%n",
+ System.out.printf(Locale.ROOT, " %s: %,d samples %,d
bytes (%.1f%% of budget)%n",
split, written, totalBytes, budgetPct);
}
}
@@ -647,14 +648,14 @@ public class BuildCharsetTrainingData {
// Summary table
System.out.println("\n=== SUMMARY ===");
- System.out.printf("%-22s %8s %12s %8s %12s %8s %12s%n",
+ System.out.printf(Locale.ROOT, "%-22s %8s %12s %8s %12s %8s %12s%n",
"Charset", "Train", "Train MB", "DevTest", "DT MB", "Test",
"Test MB");
System.out.println("-".repeat(100));
for (Map.Entry<String, Map<String, Integer>> e : manifest.entrySet()) {
String cs = e.getKey();
Map<String, Integer> sc = e.getValue();
Map<String, Long> bt = byteTotals.getOrDefault(cs,
Collections.emptyMap());
- System.out.printf("%-22s %,8d %10.1f MB %,8d %10.1f MB %,8d %10.1f
MB%n",
+ System.out.printf(Locale.ROOT, "%-22s %,8d %10.1f MB %,8d %10.1f
MB %,8d %10.1f MB%n",
cs,
sc.getOrDefault("train", 0),
bt.getOrDefault("train", 0L) / 1_000_000.0,
@@ -670,7 +671,7 @@ public class BuildCharsetTrainingData {
for (Map.Entry<String, Map<String, Integer>> e : manifest.entrySet()) {
int train = e.getValue().getOrDefault("train", 0);
if (train > 0 && train < 1000) {
- System.out.printf("WARNING: %s has only %,d train samples —
check source data!%n",
+ System.out.printf(Locale.ROOT, "WARNING: %s has only %,d train
samples — check source data!%n",
e.getKey(), train);
anyWarnings = true;
}
@@ -783,7 +784,7 @@ public class BuildCharsetTrainingData {
}
int added = result.size() - before;
if (added > 0) {
- System.out.printf(" (loaded %,d from %s)%n", added,
filename);
+ System.out.printf(Locale.ROOT, " (loaded %,d from %s)%n",
added, filename);
}
}
return result;
@@ -874,7 +875,7 @@ public class BuildCharsetTrainingData {
stopReason = "unknown";
}
if (encodeRejected > 0 || !"hit byte budget".equals(stopReason)) {
- System.out.printf(" [stop: %s | encode-rejected=%,d | "
+ System.out.printf(Locale.ROOT, " [stop: %s |
encode-rejected=%,d | "
+ "sentences-consumed=%,d/%,d]%n",
stopReason, encodeRejected, sentIdx, sentences.size());
}
diff --git
a/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/DiagnoseDiscrimination.java
b/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/DiagnoseDiscrimination.java
index 29a16e517e..46c0ba7ee2 100644
---
a/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/DiagnoseDiscrimination.java
+++
b/tika-ml/tika-ml-chardetect/src/main/java/org/apache/tika/ml/chardetect/tools/DiagnoseDiscrimination.java
@@ -386,7 +386,7 @@ public final class DiagnoseDiscrimination {
for (int i = 0; i < s.length(); ) {
int cp = s.codePointAt(i);
if (cp < 0x20 || cp == 0x7F) {
- sb.append(String.format("\\x%02X", cp));
+ sb.append(String.format(Locale.ROOT, "\\x%02X", cp));
} else if (cp == 0xFFFD) {
sb.append("<FFFD>");
} else {
diff --git a/tika-ml/tika-ml-junkdetect-tools/pom.xml
b/tika-ml/tika-ml-junkdetect-tools/pom.xml
index fe720dafe8..694377939f 100644
--- a/tika-ml/tika-ml-junkdetect-tools/pom.xml
+++ b/tika-ml/tika-ml-junkdetect-tools/pom.xml
@@ -83,13 +83,6 @@
</configuration>
</plugin>
<!-- Tools package uses System.out/printf freely -->
- <plugin>
- <groupId>de.thetaphi</groupId>
- <artifactId>forbiddenapis</artifactId>
- <configuration>
- <skip>true</skip>
- </configuration>
- </plugin>
</plugins>
</build>
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
index 4936e70886..df26127d58 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BoundaryBigramAudit.java
@@ -64,7 +64,7 @@ public final class BoundaryBigramAudit {
.sorted().toArray(Path[]::new);
}
- System.out.printf("%-22s %14s %14s %14s %14s %12s | %14s %14s%n",
+ System.out.printf(Locale.ROOT, "%-22s %14s %14s %14s %14s %12s | %14s
%14s%n",
"script", "in_S_occ", "boundary_occ", "foreign_occ",
"ascii_run_occ", "total_occ",
"drop_foreign_dist", "drop_asciirun_dist");
@@ -135,7 +135,7 @@ public final class BoundaryBigramAudit {
int distForeignDrop = distinctKeptUnderForeignDrop.size();
int distAsciiDrop = distinctKeptUnderAsciiDrop.size();
- System.out.printf("%-22s %,14d %,14d %,14d %,14d %,12d | %,14d
%,14d%n",
+ System.out.printf(Locale.ROOT, "%-22s %,14d %,14d %,14d %,14d
%,12d | %,14d %,14d%n",
name.toLowerCase(Locale.ROOT), inS, boundary, foreign,
asciiRun, total,
distAll - distForeignDrop, distAll - distAsciiDrop);
}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
index 8806be34aa..75b0a15ac1 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/BuildJunkTrainingData.java
@@ -159,16 +159,16 @@ public class BuildJunkTrainingData {
System.out.println(" data-dir: " + dataDir);
System.out.println(" output-dir: " + outputDir);
System.out.println(" --- config (JunkDetectorTrainingConfig) ---");
- System.out.printf( " total-budget-bytes: %,d (%.1f MB)%n",
+ System.out.printf(Locale.ROOT, " total-budget-bytes: %,d (%.1f
MB)%n",
totalBudgetBytes, totalBudgetBytes / 1_000_000.0);
- System.out.printf( " per-language-cap: %,d (%.1f MB)%n",
+ System.out.printf(Locale.ROOT, " per-language-cap: %,d (%.1f
MB)%n",
perLanguageCapBytes, perLanguageCapBytes / 1_000_000.0);
- System.out.printf( " min-bytes: %d%n", minBytes);
- System.out.printf( " max-punc-frac: %.2f%n", maxPuncFrac);
- System.out.printf( " min-target-script-frac: %.2f%n",
minTargetScriptFrac);
- System.out.printf( " min-dev-sentences: %d (min total ≈ %d)%n",
+ System.out.printf(Locale.ROOT, " min-bytes: %d%n",
minBytes);
+ System.out.printf(Locale.ROOT, " max-punc-frac: %.2f%n",
maxPuncFrac);
+ System.out.printf(Locale.ROOT, " min-target-script-frac: %.2f%n",
minTargetScriptFrac);
+ System.out.printf(Locale.ROOT, " min-dev-sentences: %d (min
total ≈ %d)%n",
minDevSentences, (int)(minDevSentences / DEV_FRAC));
- System.out.printf( " seed: %d%n", seed);
+ System.out.printf(Locale.ROOT, " seed: %d%n",
seed);
if (!dropScripts.isEmpty()) {
System.out.println(" drop-scripts: " + dropScripts);
}
@@ -198,19 +198,19 @@ public class BuildJunkTrainingData {
String script = detectDominantScript(langDir,
scriptSampleLines);
langToScript.put(lang, script);
scriptGroups.computeIfAbsent(script, k -> new
ArrayList<>()).add(langDir);
- System.out.printf(" %-12s → %s%n", lang, script);
+ System.out.printf(Locale.ROOT, " %-12s → %s%n", lang, script);
}
}
if (!dropScripts.isEmpty()) {
for (String s : dropScripts) {
if (scriptGroups.remove(s) != null) {
- System.out.printf(" DROP script: %s%n", s);
+ System.out.printf(Locale.ROOT, " DROP script: %s%n", s);
}
}
}
- System.out.printf("%n → %d languages, %d script groups%n",
+ System.out.printf(Locale.ROOT, "%n → %d languages, %d script
groups%n",
langToScript.size(), scriptGroups.size());
//
-----------------------------------------------------------------------
@@ -233,7 +233,7 @@ public class BuildJunkTrainingData {
double entropy = computeBigramEntropy(sample);
scriptEntropy.put(script, entropy);
- System.out.printf(" %-20s H=%.3f bits (%d sentences)%n",
+ System.out.printf(Locale.ROOT, " %-20s H=%.3f bits (%d
sentences)%n",
script, entropy, sample.size());
}
@@ -251,13 +251,13 @@ public class BuildJunkTrainingData {
long budget = (long) (totalBudgetBytes * e.getValue() /
totalEntropy);
Long override = scriptBudgetOverrides.get(e.getKey());
if (override != null) {
- System.out.printf(" %-20s H=%.3f → %,d bytes (%.1f MB)"
+ System.out.printf(Locale.ROOT, " %-20s H=%.3f → %,d bytes
(%.1f MB)"
+ " [OVERRIDE: was %,d (%.1f MB)]%n",
e.getKey(), e.getValue(), override, override /
1_000_000.0,
budget, budget / 1_000_000.0);
budget = override;
} else {
- System.out.printf(" %-20s H=%.3f → %,d bytes (%.1f MB)%n",
+ System.out.printf(Locale.ROOT, " %-20s H=%.3f → %,d bytes
(%.1f MB)%n",
e.getKey(), e.getValue(), budget, budget /
1_000_000.0);
}
scriptBudget.put(e.getKey(), budget);
@@ -265,7 +265,7 @@ public class BuildJunkTrainingData {
// Warn about overrides for scripts that aren't in the bucket set.
for (String k : scriptBudgetOverrides.keySet()) {
if (!scriptBudget.containsKey(k)) {
- System.err.printf("WARNING: --script-budget-override for %s
ignored"
+ System.err.printf(Locale.ROOT, "WARNING:
--script-budget-override for %s ignored"
+ " (script not in bucket set)%n", k);
}
}
@@ -315,7 +315,7 @@ public class BuildJunkTrainingData {
sentences);
totalBytesLoaded += langBytes;
if (langBytes > 0) {
- System.out.printf(" %-12s %-20s +%,d bytes%n",
+ System.out.printf(Locale.ROOT, " %-12s %-20s +%,d
bytes%n",
script, langDir.getFileName(), langBytes);
}
}
@@ -335,7 +335,7 @@ public class BuildJunkTrainingData {
// Round 2: redistribute surplus to saturated scripts proportional to
entropy
if (surplus > 0) {
- System.out.printf(
+ System.out.printf(Locale.ROOT,
"\n--- Phase 4b: Redistributing %,d surplus bytes (%.1f
MB) ---\n",
surplus, surplus / 1_000_000.0);
@@ -377,7 +377,7 @@ public class BuildJunkTrainingData {
if (!sentences.isEmpty()) {
allSentences.put(script, sentences);
actualBytes.put(script, totalBytesLoaded);
- System.out.printf(" %-20s +%,d extra → %,d total bytes,
%,d sentences%n",
+ System.out.printf(Locale.ROOT, " %-20s +%,d extra → %,d
total bytes, %,d sentences%n",
script, extra, totalBytesLoaded, sentences.size());
}
}
@@ -395,7 +395,7 @@ public class BuildJunkTrainingData {
int expectedDevSize = (int) (sentences.size() * DEV_FRAC);
if (sentences.isEmpty() || expectedDevSize < minDevSentences) {
- System.out.printf(
+ System.out.printf(Locale.ROOT,
" SKIP %-20s — %,d sentences → dev=%d <
min-dev-sentences=%d%n",
script, sentences.size(), expectedDevSize,
minDevSentences);
manifestStats.put(script, new long[]{0, 0, 0, 0, 0});
@@ -418,7 +418,7 @@ public class BuildJunkTrainingData {
long totalBytesLoaded = actualBytes.getOrDefault(script, 0L);
manifestStats.put(script,
new long[]{totalBytesLoaded, sentences.size(), nTrain,
nDev, test.size()});
- System.out.printf(
+ System.out.printf(Locale.ROOT,
" WROTE %-12s — %,d bytes, %,d sentences (train=%,d
dev=%,d test=%,d)%n",
script, totalBytesLoaded, sentences.size(),
nTrain, nDev, test.size());
@@ -440,7 +440,7 @@ public class BuildJunkTrainingData {
String langs = scriptGroups.get(script).stream()
.map(p -> p.getFileName().toString())
.reduce((a, b) -> a + "," + b).orElse("");
- w.write(String.format("%s\t%.3f\t%d\t%d\t%d\t%d\t%d\t%d\t%s%n",
+ w.write(String.format(Locale.ROOT,
"%s\t%.3f\t%d\t%d\t%d\t%d\t%d\t%d\t%s%n",
script, entropy, budget,
stats[0], stats[1], stats[2], stats[3], stats[4],
langs));
}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/DebugScriptRuns.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/DebugScriptRuns.java
index 36f3a897a0..8044c3b576 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/DebugScriptRuns.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/DebugScriptRuns.java
@@ -25,6 +25,7 @@ import java.nio.file.Paths;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.List;
+import java.util.Locale;
import java.util.Map;
import java.util.TreeMap;
import java.util.regex.Matcher;
@@ -139,7 +140,7 @@ public class DebugScriptRuns {
System.out.println("Script roll-up (script: cps, utf8_bytes, runs,
modeled):");
for (Map.Entry<String, int[]> e : totals.entrySet()) {
int[] v = e.getValue();
- System.out.printf(" %-15s cps=%-5d bytes=%-6d runs=%-4d
modeled=%s%n",
+ System.out.printf(Locale.ROOT, " %-15s cps=%-5d bytes=%-6d
runs=%-4d modeled=%s%n",
e.getKey(), v[0], v[1], v[2], v[3] == 1 ? "Y" : "N");
}
System.out.println();
@@ -166,7 +167,7 @@ public class DebugScriptRuns {
TextQualityScore score = detector.score(decoded);
System.out.println(" detector.score() z: "
+ (score.isUnknown() ? "UNKNOWN(" + score.getDominantScript()
+ ")"
- : String.format("%.3f (script=%s)", score.getZScore(),
score.getDominantScript())));
+ : String.format(Locale.ROOT, "%.3f (script=%s)",
score.getZScore(), score.getDominantScript())));
// Print the longest 10 runs so we can see what's actually in there.
System.out.println();
@@ -178,7 +179,7 @@ public class DebugScriptRuns {
String preview = r.text.length() > 30
? r.text.substring(0, 30) + "…" : r.text;
preview = preview.replace("\n", "\\n").replace("\r", "\\r");
- System.out.printf(" %-15s cps=%-4d bytes=%-4d preview=%s%n",
+ System.out.printf(Locale.ROOT, " %-15s cps=%-4d bytes=%-4d
preview=%s%n",
r.script, r.text.codePointCount(0, r.text.length()),
u.length, preview);
}
}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
index 21cdefa954..ba111cd3c2 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/LineScriptFractions.java
@@ -68,7 +68,7 @@ public final class LineScriptFractions {
System.exit(1);
}
- System.out.printf("%-20s %10s %10s | %s%n",
+ System.out.printf(Locale.ROOT, "%-20s %10s %10s | %s%n",
"script", "lines", "<5%",
"lines at target-frac threshold (cumulative dropped %)");
System.out.println(" "
@@ -81,7 +81,7 @@ public final class LineScriptFractions {
.toUpperCase(Locale.ROOT);
Character.UnicodeScript target = mapScript(name);
if (target == null) {
- System.out.printf("%-20s (no UnicodeScript mapping for
'%s')%n", name, name);
+ System.out.printf(Locale.ROOT, "%-20s (no UnicodeScript
mapping for '%s')%n", name, name);
continue;
}
@@ -131,11 +131,11 @@ public final class LineScriptFractions {
if (hi <= t) dropped += bucketCounts[j];
}
double pct = 100.0 * dropped / Math.max(1, lines);
- sb.append(String.format(" %6.1f", pct));
+ sb.append(String.format(Locale.ROOT, " %6.1f", pct));
}
long below5 = bucketCounts[0];
- System.out.printf("%-20s %,10d %,10d |%s%n",
+ System.out.printf(Locale.ROOT, "%-20s %,10d %,10d |%s%n",
name.toLowerCase(Locale.ROOT), lines, below5,
sb.toString());
}
}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/ScriptCensus.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/ScriptCensus.java
index b384d5f4c5..c7a0421113 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/ScriptCensus.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/ScriptCensus.java
@@ -27,6 +27,7 @@ import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashMap;
import java.util.List;
+import java.util.Locale;
import java.util.Map;
import java.util.zip.GZIPInputStream;
@@ -116,8 +117,8 @@ public final class ScriptCensus {
}
}
- System.out.printf("File: %s%n", file);
- System.out.printf(" lines sampled: %,d total codepoints (excl.
COMMON/INHERITED): %,d%n%n",
+ System.out.printf(Locale.ROOT, "File: %s%n", file);
+ System.out.printf(Locale.ROOT, " lines sampled: %,d total
codepoints (excl. COMMON/INHERITED): %,d%n%n",
lines, total);
if (total == 0) {
@@ -135,7 +136,7 @@ public final class ScriptCensus {
double pct = 100.0 * c / total;
double cumPct = 100.0 * cumulative / total;
if (pct < 0.01 && c < 100) continue;
- System.out.printf(" %-22s %,14d %6.2f%% (cum %6.2f%%)%n",
+ System.out.printf(Locale.ROOT, " %-22s %,14d %6.2f%% (cum
%6.2f%%)%n",
e.getKey(), c, pct, cumPct);
}
@@ -149,7 +150,7 @@ public final class ScriptCensus {
long c = e.getValue()[0];
double pct = 100.0 * c / domTotal;
if (pct < 0.05) continue;
- System.out.printf(" %-22s %,12d %6.2f%% of lines%n",
+ System.out.printf(Locale.ROOT, " %-22s %,12d %6.2f%% of
lines%n",
e.getKey(), c, pct);
}
}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
index 4bebad6fa3..33629e97f2 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/main/java/org/apache/tika/ml/junkdetect/tools/TrainJunkModel.java
@@ -238,11 +238,11 @@ public class TrainJunkModel {
System.out.println(" data-dir: " + dataDir);
System.out.println(" output: " + output);
System.out.println(" --- format constants (TrainJunkModel) ---");
- System.out.printf( " backoff_alpha: %.2f%n", BACKOFF_ALPHA);
+ System.out.printf(Locale.ROOT, " backoff_alpha: %.2f%n",
BACKOFF_ALPHA);
System.out.println(" --- config (JunkDetectorTrainingConfig) ---");
- System.out.printf( " min_bigram_count: %d%n", minBigramCount);
- System.out.printf( " oa_load_factor: %.2f%n", loadFactor);
- System.out.printf( " key_index_bits: %d%n", keyIndexBits);
+ System.out.printf(Locale.ROOT, " min_bigram_count: %d%n",
minBigramCount);
+ System.out.printf(Locale.ROOT, " oa_load_factor: %.2f%n",
loadFactor);
+ System.out.printf(Locale.ROOT, " key_index_bits: %d%n",
keyIndexBits);
if (!Files.isDirectory(dataDir)) {
System.err.println("ERROR: data-dir not found: " + dataDir);
@@ -250,7 +250,7 @@ public class TrainJunkModel {
}
int blockN =
org.apache.tika.ml.junkdetect.UnicodeBlockRanges.bucketCount();
- System.out.printf("Block bucketing: %d named blocks + 1 unassigned "
+ System.out.printf(Locale.ROOT, "Block bucketing: %d named blocks + 1
unassigned "
+ "(scheme version %d, JVM-independent)%n",
blockN - 1,
org.apache.tika.ml.junkdetect.UnicodeBlockRanges.SCHEME_VERSION);
long t0 = System.currentTimeMillis();
@@ -303,7 +303,7 @@ public class TrainJunkModel {
unigramsByScript.get(script), totals[0],
minBigramCount, loadFactor, keyIndexBits);
f1TablesByScript.put(script, tables);
- System.out.printf(" [%s] %s (%dms)%n", script,
tables.statsString(),
+ System.out.printf(Locale.ROOT, " [%s] %s (%dms)%n", script,
tables.statsString(),
System.currentTimeMillis() - t0);
}
@@ -331,7 +331,7 @@ public class TrainJunkModel {
float[] cal = scores.isEmpty() ? new float[]{0f, 1f} :
muSigma(scores);
cal[1] = Math.max(cal[1], Z1_MIN_SIGMA); // floor degenerate
(under-trained) sigma
f1Calibrations.put(script, cal);
- System.out.printf(" [%s] mu=%.4f sigma=%.4f (%,d windows)%n",
+ System.out.printf(Locale.ROOT, " [%s] mu=%.4f sigma=%.4f (%,d
windows)%n",
script, cal[0], cal[1], scores.size());
}
@@ -350,15 +350,15 @@ public class TrainJunkModel {
scriptTransTable = quantizeDequantizeRoundTrip(scriptTransTable);
float[] scriptTransCal = calibrateScriptTransitions(allTrainFiles,
scriptTransTable,
scriptBucketMap, numScriptBuckets);
- System.out.printf(" scriptTrans: mu=%.4f sigma=%.4f%n",
+ System.out.printf(Locale.ROOT, " scriptTrans: mu=%.4f sigma=%.4f%n",
scriptTransCal[0], scriptTransCal[1]);
float[] blockTable =
quantizeDequantizeRoundTrip(trainGlobalBlockTable(allTrainFiles));
float[] blockCal = computeGlobalBlockCalibration(allTrainFiles,
blockTable);
- System.out.printf(" block: mu=%.4f sigma=%.4f%n", blockCal[0],
blockCal[1]);
+ System.out.printf(Locale.ROOT, " block: mu=%.4f sigma=%.4f%n",
blockCal[0], blockCal[1]);
float[] controlCal = computeGlobalControlCalibration(allTrainFiles);
- System.out.printf(" control: mu=%.6f sigma=%.6f%n",
controlCal[0], controlCal[1]);
+ System.out.printf(Locale.ROOT, " control: mu=%.6f sigma=%.6f%n",
controlCal[0], controlCal[1]);
//
-----------------------------------------------------------------------
// Phase 3 — ONE global combiner over z1..z9, trained pointwise
@@ -376,17 +376,17 @@ public class TrainJunkModel {
t0 = System.currentTimeMillis();
float[] combiner = trainGlobalCombiner(featExtractor, trainFilePaths);
- System.out.printf(
+ System.out.printf(Locale.ROOT,
" global w=[%.3f,%.3f,%.3f,%.3f,%.3f,%.3f,%.3f,%.3f,%.3f]
bias=%.3f (%dms)%n",
combiner[0], combiner[1], combiner[2], combiner[3],
combiner[4], combiner[5],
combiner[6], combiner[7], combiner[8], combiner[9],
System.currentTimeMillis() - t0);
- System.out.printf("%nWriting model (%d scripts, blockN=%d,
scriptBuckets=%d) → %s%n",
+ System.out.printf(Locale.ROOT, "%nWriting model (%d scripts,
blockN=%d, scriptBuckets=%d) → %s%n",
f1TablesByScript.size(), blockN, numScriptBuckets, output);
saveModel(f1TablesByScript, f1Calibrations, blockTable, blockCal,
controlCal,
combiner, scriptBuckets, scriptTransTable, scriptTransCal,
output);
- System.out.printf("Model size: %,d bytes (%.1f KB)%n",
+ System.out.printf(Locale.ROOT, "Model size: %,d bytes (%.1f KB)%n",
Files.size(output), Files.size(output) / 1024.0);
System.out.println("Done.");
}
@@ -1081,7 +1081,7 @@ public class TrainJunkModel {
values[i] = (byte) (sortable[i] & 0xFF);
}
- System.out.printf(
+ System.out.printf(Locale.ROOT,
" pair_counts: distinct=%,d, kept=%,d (>=%d), dropped=%,d "
+ "cp_index=%,d bigram_entries=%,d%n",
totalDistinct, keptPairs, minBigramCount, dropped,
@@ -1218,7 +1218,7 @@ public class TrainJunkModel {
}
}
}
- System.out.printf(" examples: good=%,d bad=%,d pairs=%,d%n",
+ System.out.printf(Locale.ROOT, " examples: good=%,d bad=%,d
pairs=%,d%n",
good.size(), bad.size(), pairCorrect.size());
return fitContrastiveCombiner(good, bad, pairCorrect, pairWrong);
}
@@ -1612,7 +1612,7 @@ public class TrainJunkModel {
}
}
}
- System.out.printf("%,d script transitions across %d files%n",
totalTransitions, trainFiles.size());
+ System.out.printf(Locale.ROOT, "%,d script transitions across %d
files%n", totalTransitions, trainFiles.size());
return laplaceSmoothLogProb(counts, numBuckets);
}
@@ -1638,7 +1638,7 @@ public class TrainJunkModel {
}
}
}
- System.out.printf("%,d dev windows pooled%n", scores.size());
+ System.out.printf(Locale.ROOT, "%,d dev windows pooled%n",
scores.size());
return muSigma(scores);
}
diff --git
a/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
b/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
index c135f9357e..d6f1ed5a82 100644
---
a/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
+++
b/tika-ml/tika-ml-junkdetect-tools/src/test/java/org/apache/tika/ml/junkdetect/tools/BuildJunkAugmentationData.java
@@ -271,14 +271,14 @@ public final class BuildJunkAugmentationData {
? null
: loadProfileCsv(profileCsv);
if (profiles != null) {
- System.out.printf(" loaded %,d profile rows%n", profiles.size());
+ System.out.printf(Locale.ROOT, " loaded %,d profile rows%n",
profiles.size());
}
// --- Phase 1: discover baseline scripts + line counts
-------------------
System.out.println("\n--- Phase 1: scanning baseline train files ---");
Map<String, Long> baselineLineCounts =
scanBaselineLineCounts(baselineDir);
for (Map.Entry<String, Long> e : baselineLineCounts.entrySet()) {
- System.out.printf(" %-20s baseline=%,d lines%n", e.getKey(),
e.getValue());
+ System.out.printf(Locale.ROOT, " %-20s baseline=%,d lines%n",
e.getKey(), e.getValue());
}
// --- Phase 2: walk extracts
---------------------------------------------
@@ -369,17 +369,17 @@ public final class BuildJunkAugmentationData {
}
}
- System.out.printf(" total extracts seen: %,d%n", totalSeen);
- System.out.printf(" dropped no-content: %,d%n", droppedNoContent);
- System.out.printf(" dropped short: %,d%n", droppedShort);
+ System.out.printf(Locale.ROOT, " total extracts seen: %,d%n",
totalSeen);
+ System.out.printf(Locale.ROOT, " dropped no-content: %,d%n",
droppedNoContent);
+ System.out.printf(Locale.ROOT, " dropped short: %,d%n",
droppedShort);
if (profiles != null) {
- System.out.printf(" dropped no-profile: %,d%n",
droppedNoProfile);
- System.out.printf(" dropped OOV>%.2f: %,d%n", maxOov,
droppedOov);
- System.out.printf(" dropped langness<%.2f: %,d%n", minLangness,
droppedLangness);
+ System.out.printf(Locale.ROOT, " dropped no-profile: %,d%n",
droppedNoProfile);
+ System.out.printf(Locale.ROOT, " dropped OOV>%.2f: %,d%n",
maxOov, droppedOov);
+ System.out.printf(Locale.ROOT, " dropped langness<%.2f: %,d%n",
minLangness, droppedLangness);
}
- System.out.printf(" dropped mixed-script: %,d%n",
droppedMixedScript);
- System.out.printf(" dropped no-baseline: %,d%n",
droppedNoBaseline);
- System.out.printf(" contributed ≥1 chunk: %,d%n", accepted);
+ System.out.printf(Locale.ROOT, " dropped mixed-script: %,d%n",
droppedMixedScript);
+ System.out.printf(Locale.ROOT, " dropped no-baseline: %,d%n",
droppedNoBaseline);
+ System.out.printf(Locale.ROOT, " contributed ≥1 chunk: %,d%n",
accepted);
// --- Phase 3: apply per-script gates and caps
---------------------------
System.out.println("\n--- Phase 3: per-script gating and capping ---");
@@ -398,7 +398,7 @@ public final class BuildJunkAugmentationData {
long cap = Math.min(hardCap, fracCapVal);
if (docs < minDocs) {
- System.out.printf(
+ System.out.printf(Locale.ROOT,
" SKIP %-20s docs=%-6d (<%d gate) chunks=%,d
cap=%,d%n",
script, docs, minDocs, chunks.size(), cap);
manifest.put(script, new long[]{docs, chunks.size(), 0,
baselineLines, cap});
@@ -428,7 +428,7 @@ public final class BuildJunkAugmentationData {
for (int k = quota; kept.size() < cap && k < withSym.size();
k++) {
kept.add(withSym.get(k));
}
- System.out.printf(
+ System.out.printf(Locale.ROOT,
" KEEP %-20s docs=%-6d chunks=%,8d -> append=%,6d "
+ "(baseline=%,d, cap=%,d, symbol-quota=%d,
symbol-bearing-pool=%d)%n",
script, docs, chunks.size(), kept.size(),
baselineLines, cap,
@@ -437,7 +437,7 @@ public final class BuildJunkAugmentationData {
kept = chunks.size() > cap
? new ArrayList<>(chunks.subList(0, (int) cap))
: chunks;
- System.out.printf(
+ System.out.printf(Locale.ROOT,
" KEEP %-20s docs=%-6d chunks=%,8d -> append=%,6d
(baseline=%,d, cap=%,d)%n",
script, docs, chunks.size(), kept.size(),
baselineLines, cap);
}
@@ -468,10 +468,10 @@ public final class BuildJunkAugmentationData {
List<String> add = finalLines.get(script);
if (add != null && !add.isEmpty()) {
rewriteTrainWithAppend(src, dst, add);
- System.out.printf(" WROTE %-30s +%,d lines
appended%n", name, add.size());
+ System.out.printf(Locale.ROOT, " WROTE %-30s +%,d
lines appended%n", name, add.size());
} else {
Files.copy(src, dst);
- System.out.printf(" COPY %-30s (no
augmentation)%n", name);
+ System.out.printf(Locale.ROOT, " COPY %-30s (no
augmentation)%n", name);
}
} else {
Files.copy(src, dst);
@@ -485,7 +485,7 @@ public final class BuildJunkAugmentationData {
w.write("script\tdocs\tchunks_pre_cap\tlines_appended\tbaseline_lines\tcap\n");
for (Map.Entry<String, long[]> e : manifest.entrySet()) {
long[] r = e.getValue();
- w.write(String.format("%s\t%d\t%d\t%d\t%d\t%d%n",
+ w.write(String.format(Locale.ROOT, "%s\t%d\t%d\t%d\t%d\t%d%n",
e.getKey(), r[0], r[1], r[2], r[3], r[4]));
}
}
diff --git a/tika-ml/tika-ml-junkdetect/pom.xml
b/tika-ml/tika-ml-junkdetect/pom.xml
index af9e7c456f..ae30c8247d 100644
--- a/tika-ml/tika-ml-junkdetect/pom.xml
+++ b/tika-ml/tika-ml-junkdetect/pom.xml
@@ -97,13 +97,6 @@
</configuration>
</plugin>
<!-- Diagnostic tests print to stdout / use default-locale formatting
freely. -->
- <plugin>
- <groupId>de.thetaphi</groupId>
- <artifactId>forbiddenapis</artifactId>
- <configuration>
- <skip>true</skip>
- </configuration>
- </plugin>
</plugins>
</build>
diff --git
a/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/LatinSiblingComparisonTest.java
b/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/LatinSiblingComparisonTest.java
index 8965d841fe..57b7fb2632 100644
---
a/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/LatinSiblingComparisonTest.java
+++
b/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/LatinSiblingComparisonTest.java
@@ -21,6 +21,7 @@ import static org.junit.jupiter.api.Assertions.assertEquals;
import java.nio.charset.Charset;
import java.util.ArrayList;
import java.util.List;
+import java.util.Locale;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
@@ -108,11 +109,11 @@ public class LatinSiblingComparisonTest {
TextQualityComparison cmp = detector.compare(
"windows-1252", asWin1252, wrong, asWrong);
- String tag = String.format("%-20s vs %-12s", probe.name,
wrong);
+ String tag = String.format(Locale.ROOT, "%-20s vs %-12s",
probe.name, wrong);
if ("windows-1252".equals(cmp.winner())) {
- passes.add(String.format("PASS %s delta=%.3f", tag,
cmp.delta()));
+ passes.add(String.format(Locale.ROOT, "PASS %s
delta=%.3f", tag, cmp.delta()));
} else {
- failures.add(String.format("FAIL %s winner=%-12s
delta=%.3f",
+ failures.add(String.format(Locale.ROOT, "FAIL %s
winner=%-12s delta=%.3f",
tag, cmp.winner(), cmp.delta()));
}
}
@@ -121,7 +122,7 @@ public class LatinSiblingComparisonTest {
System.out.println("\n=== Latin SBCS sibling comparison: " + label + "
===");
passes.forEach(System.out::println);
failures.forEach(System.out::println);
- System.out.printf("%d pass, %d fail (of %d cells)%n",
+ System.out.printf(Locale.ROOT, "%d pass, %d fail (of %d cells)%n",
passes.size(), failures.size(), passes.size() +
failures.size());
assertEquals(0, failures.size(),
diff --git a/tika-parent/checkstyle.xml b/tika-parent/checkstyle.xml
index cf9928edf4..72c5050e1f 100644
--- a/tika-parent/checkstyle.xml
+++ b/tika-parent/checkstyle.xml
@@ -65,8 +65,8 @@
<!--<module name="FileContentsHolder"/>-->
<module name="IllegalImport">
<property name="regexp" value="true"/>
- <!-- Reject any org.junit import that's not also org.junit.jupiter: -->
- <property name="illegalClasses" value="^org\.junit\.(?!jupiter\.).+"/>
+ <!-- Reject JUnit 4: any org.junit import outside org.junit.jupiter and
org.junit.platform -->
+ <property name="illegalClasses"
value="^org\.junit\.(?!(jupiter|platform)\.).+"/>
</module>
<module name="OuterTypeFilename"/>
<module name="IllegalTokenText">
diff --git a/tika-parent/pom.xml b/tika-parent/pom.xml
index 4e6e363b54..edd2d7335c 100644
--- a/tika-parent/pom.xml
+++ b/tika-parent/pom.xml
@@ -1522,8 +1522,9 @@
<artifactId>maven-surefire-plugin</artifactId>
<version>${maven.surefire.version}</version>
<configuration>
- <!-- for manual testing of i18n, try for example: -Duser.language=zh
-Duser.region=CN or
- -Duser.language=de -Duser.country=DE -->
+ <!-- -Dtika.test.locale=random (CI) or =tr-TR runs the tests under
that default locale
+ (RandomLocaleListener). -Duser.language here would not work:
surefire sets
+ it in the fork only after the default Locale is fixed. -->
<!-- java.io.tmpdir MUST be quoted: argLine is split on whitespace,
and the Windows
job deliberately checks out into a path containing a space. It
must also be a
real -D on the command line, not <systemPropertyVariables>:
Files.createTempFile
@@ -1734,6 +1735,28 @@
</build>
<profiles>
+ <profile>
+ <!-- Puts RandomLocaleListener on the test classpath of every module
that has tests:
+ each test JVM then runs under a random default locale. A profile,
not a plain
+ dependency: aggregator poms and tika-test-support itself (which
keeps its test
+ in tika-core for this reason) must not depend on it, or the reactor
cycles. -->
+ <id>test-support</id>
+ <activation>
+ <file>
+ <exists>${basedir}/src/test/java</exists>
+ </file>
+ </activation>
+ <dependencies>
+ <dependency>
+ <groupId>org.apache.tika</groupId>
+ <artifactId>tika-test-support</artifactId>
+ <!-- revision, not project.version: flatten writes the literal Tika
version
+ into the published tika-parent, so a foreign child of it
resolves the right jar -->
+ <version>${revision}</version>
+ <scope>test</scope>
+ </dependency>
+ </dependencies>
+ </profile>
<profile>
<id>pedantic</id>
<build>
diff --git
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/grib/GribParserTest.java
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/grib/GribParserTest.java
index 7768eda1fc..57d5806874 100644
---
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/grib/GribParserTest.java
+++
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/grib/GribParserTest.java
@@ -18,6 +18,10 @@ package org.apache.tika.parser.grib;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
+import static org.junit.jupiter.api.Assumptions.assumeTrue;
+
+import java.text.DecimalFormatSymbols;
+import java.util.Locale;
import org.junit.jupiter.api.Test;
import org.xml.sax.ContentHandler;
@@ -37,6 +41,10 @@ public class GribParserTest {
@Test
public void testParseGlobalMetadata() throws Exception {
Parser parser = new GribParser();
+ // TODO: netcdf-java formats with the default locale, so dimension
sizes and a WMO
+ // code-table file name come out in the locale's digits; report
upstream
+
assumeTrue(DecimalFormatSymbols.getInstance(Locale.getDefault()).getZeroDigit()
== '0',
+ "netcdf-java needs ASCII digits in the default locale");
Metadata metadata = new Metadata();
ContentHandler handler = new BodyContentHandler();
try (TikaInputStream tis = TikaInputStream.get(GribParser.class
diff --git
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/netcdf/NetCDFParserTest.java
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/netcdf/NetCDFParserTest.java
index 84b8fb0ffb..0e031dce37 100644
---
a/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/netcdf/NetCDFParserTest.java
+++
b/tika-parsers/tika-parsers-extended/tika-parser-scientific-module/src/test/java/org/apache/tika/parser/netcdf/NetCDFParserTest.java
@@ -18,6 +18,10 @@ package org.apache.tika.parser.netcdf;
import static org.apache.tika.TikaTest.assertContains;
import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assumptions.assumeTrue;
+
+import java.text.DecimalFormatSymbols;
+import java.util.Locale;
import org.junit.jupiter.api.Test;
import org.xml.sax.ContentHandler;
@@ -38,6 +42,10 @@ public class NetCDFParserTest {
@Test
public void testParseGlobalMetadata() throws Exception {
Parser parser = new NetCDFParser();
+ // TODO: netcdf-java formats with the default locale, so dimension
sizes and a WMO
+ // code-table file name come out in the locale's digits; report
upstream
+
assumeTrue(DecimalFormatSymbols.getInstance(Locale.getDefault()).getZeroDigit()
== '0',
+ "netcdf-java needs ASCII digits in the default locale");
ContentHandler handler = new BodyContentHandler();
Metadata metadata = new Metadata();
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
index 9434b43971..f3ee0a83a5 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-integration-tests/src/test/java/org/apache/tika/parser/ocr/TesseractOCRParserTest.java
@@ -258,8 +258,8 @@ public class TesseractOCRParserTest extends TikaTest {
assertEquals("75", m.get(TIFF.IMAGE_LENGTH));
}
- // TODO TIKA-4923: metadata-extractor lowercases the resolution unit in
the default locale
- @DisabledIfSystemProperty(named = "user.language", matches = "tr")
+ // TODO TIKA-4923: metadata-extractor lowercases the resolution unit in
the default locale (tr/az dotless i)
+ @DisabledIfSystemProperty(named = "tika.test.locale", matches =
"(tr|az)(-.*)?")
@Test
public void getNormalMetadataTooUnknownField() throws Exception {
Metadata m = getXML("testTIFF.tif").metadata;
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/sas/SAS7BDATParser.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/sas/SAS7BDATParser.java
index dfa4a6099c..79c3954719 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/sas/SAS7BDATParser.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/sas/SAS7BDATParser.java
@@ -20,6 +20,7 @@ import java.io.IOException;
import java.text.Format;
import java.util.Collections;
import java.util.HashMap;
+import java.util.Locale;
import java.util.Map;
import java.util.Set;
@@ -141,7 +142,8 @@ public class SAS7BDATParser implements Parser {
Object[] row = null;
while ((row = sas.readNext()) != null) {
xhtml.startElement("tr");
- for (String val : DataWriterUtil.getRowValues(sas.getColumns(),
row, formatMap)) {
+ // parso otherwise formats dates and percents with the JVM default
locale
+ for (String val : DataWriterUtil.getRowValues(sas.getColumns(),
row, Locale.US, formatMap)) {
// Use explicit start/end, rather than element, to
// ensure that empty cells still get output
xhtml.startElement("td");
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/test/java/org/apache/tika/parser/sas/SAS7BDATParserTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/test/java/org/apache/tika/parser/sas/SAS7BDATParserTest.java
index 22b6d47635..50cb9af7e0 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/test/java/org/apache/tika/parser/sas/SAS7BDATParserTest.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/test/java/org/apache/tika/parser/sas/SAS7BDATParserTest.java
@@ -18,9 +18,7 @@ package org.apache.tika.parser.sas;
import static org.junit.jupiter.api.Assertions.assertEquals;
-import java.text.DateFormatSymbols;
import java.util.Arrays;
-import java.util.Locale;
import org.junit.jupiter.api.Test;
import org.xml.sax.ContentHandler;
@@ -39,8 +37,6 @@ import org.apache.tika.parser.Parser;
import org.apache.tika.sax.BodyContentHandler;
public class SAS7BDATParserTest extends TikaTest {
- private static final String[] SHORT_MONTHS =
- new DateFormatSymbols(Locale.getDefault()).getShortMonths();
private Parser parser = new SAS7BDATParser();
@Test
@@ -111,7 +107,7 @@ public class SAS7BDATParserTest extends TikaTest {
assertContains("2\t4\tThis", content);
assertContains("4\t16\tThis", content);
assertContains("\t01-01-1960\t", content);
- assertContains("\t01" + SHORT_MONTHS[0] + "1960:00:00", content);
+ assertContains("\t01Jan1960:00:00", content);
}
@Test
@@ -143,6 +139,6 @@ public class SAS7BDATParserTest extends TikaTest {
assertContains("<th title=\"date\">date</th>", xml);
// Check formatting of dates
assertContains("<td>01-01-1960</td>", xml);
- assertContains("<td>01" + SHORT_MONTHS[0] + "1960:00:00:10.00</td>",
xml);
+ assertContains("<td>01Jan1960:00:00:10.00</td>", xml);
}
}
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/main/java/org/apache/tika/parser/image/ImageMetadataExtractor.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/main/java/org/apache/tika/parser/image/ImageMetadataExtractor.java
index 0ee7e82fbc..74be1c0378 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/main/java/org/apache/tika/parser/image/ImageMetadataExtractor.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/main/java/org/apache/tika/parser/image/ImageMetadataExtractor.java
@@ -24,6 +24,10 @@ import java.nio.channels.SeekableByteChannel;
import java.text.DecimalFormat;
import java.text.DecimalFormatSymbols;
import java.text.SimpleDateFormat;
+import java.time.DateTimeException;
+import java.time.LocalDate;
+import java.time.LocalTime;
+import java.time.ZoneOffset;
import java.util.Arrays;
import java.util.Date;
import java.util.Iterator;
@@ -553,13 +557,20 @@ public class ImageMetadataExtractor {
static class ExifHandler implements DirectoryHandler {
// There's a new ExifHandler for each file processed, so this is
thread safe
- // EXIF dates have no zone: read and write in GMT (metadata-extractor
>= 2.20 uses the JVM zone)
+ // EXIF dates have no zone: write them in GMT
private static final TimeZone GMT = TimeZone.getTimeZone("GMT");
private final SimpleDateFormat dateUnspecifiedTz =
getUnspecifiedTzDateFormat();
- // metadata-extractor turns junk like "2" into year 1
- private static Date inBounds(Date d) {
- return d != null && TikaDates.inYearBounds(d.toInstant()) ? d :
null;
+ // metadata-extractor parses with the default locale's calendar;
TikaDates does not
+ private static Date exifDate(Directory directory, int tag) {
+ Object o = directory.getObject(tag);
+ if (o instanceof Date) {
+ return TikaDates.inYearBounds(((Date) o).toInstant()) ? (Date)
o : null;
+ }
+ if (o == null) {
+ return null;
+ }
+ return TikaDates.parse(o.toString()).map(d ->
Date.from(d.toInstant())).orElse(null);
}
private SimpleDateFormat getUnspecifiedTzDateFormat() {
@@ -734,7 +745,7 @@ public class ImageMetadataExtractor {
// Date/Time Original overrides value from
ExifDirectory.TAG_DATETIME
Date original = null;
if
(directory.containsTag(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)) {
- original =
inBounds(directory.getDate(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL, GMT));
+ original = exifDate(directory,
ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL);
// Unless we have GPS time we don't know the time zone so date
must be set
// as ISO 8601 datetime without timezone suffix (no Z or +/-)
if (original != null) {
@@ -744,7 +755,7 @@ public class ImageMetadataExtractor {
}
}
if (directory.containsTag(ExifIFD0Directory.TAG_DATETIME)) {
- Date datetime =
inBounds(directory.getDate(ExifIFD0Directory.TAG_DATETIME, GMT));
+ Date datetime = exifDate(directory,
ExifIFD0Directory.TAG_DATETIME);
if (datetime != null) {
String datetimeNoTimeZone =
dateUnspecifiedTz.format(datetime);
metadata.set(TikaCoreProperties.MODIFIED,
datetimeNoTimeZone);
@@ -807,6 +818,25 @@ public class ImageMetadataExtractor {
* Maps EXIF Geo Tags onto the Tika Geo metadata namespace.
*/
static class GeotagHandler implements DirectoryHandler {
+ // GpsDirectory.getGpsDate parses with the default locale's calendar
+ static Date gpsDate(GpsDirectory directory) {
+ String stamp = directory.getString(GpsDirectory.TAG_DATE_STAMP);
+ Rational[] time =
directory.getRationalArray(GpsDirectory.TAG_TIME_STAMP);
+ if (stamp == null || time == null || time.length != 3) {
+ return null;
+ }
+ try {
+ LocalDate day = TikaDates.parse(stamp).map(d ->
d.getLocalDateTime().toLocalDate()).orElse(null);
+ if (day == null) {
+ return null;
+ }
+ LocalTime t = LocalTime.of(time[0].intValue(),
time[1].intValue(), (int) time[2].doubleValue());
+ return Date.from(day.atTime(t).toInstant(ZoneOffset.UTC));
+ } catch (DateTimeException e) {
+ return null;
+ }
+ }
+
public boolean supports(Class<? extends Directory> directoryType) {
return directoryType == GpsDirectory.class;
}
@@ -821,7 +851,7 @@ public class ImageMetadataExtractor {
metadata.set(TikaCoreProperties.LONGITUDE,
geoDecimalFormat.format(geoLocation.getLongitude()));
}
- Date gpsDate = ((GpsDirectory)directory).getGpsDate();
+ Date gpsDate = gpsDate((GpsDirectory) directory);
if (gpsDate != null) {
metadata.set(Geographic.TIMESTAMP, gpsDate);
}
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorLocaleTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorLocaleTest.java
index a4e015cb25..7bb9b1b246 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorLocaleTest.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorLocaleTest.java
@@ -18,10 +18,19 @@ package org.apache.tika.parser.image;
import static org.junit.jupiter.api.Assertions.assertEquals;
+import java.io.InputStream;
import java.util.Locale;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.parallel.Isolated;
+import org.xml.sax.helpers.DefaultHandler;
+
+import org.apache.tika.io.TikaInputStream;
+import org.apache.tika.metadata.Geographic;
+import org.apache.tika.metadata.Metadata;
+import org.apache.tika.metadata.TIFF;
+import org.apache.tika.metadata.TikaCoreProperties;
+import org.apache.tika.parser.ParseContext;
// sets the JVM-wide default locale, so it must not overlap other test classes
@Isolated
@@ -38,4 +47,31 @@ public class ImageMetadataExtractorLocaleTest {
Locale.setDefault(defaultLocale);
}
}
+
+ // the Japanese imperial calendar makes a default-locale SimpleDateFormat
read 2009 as Reiwa 2009
+ @Test
+ public void testExifDatesIgnoreDefaultCalendar() throws Exception {
+ Locale defaultLocale = Locale.getDefault();
+ try {
+ Locale.setDefault(Locale.forLanguageTag("ja-JP-u-ca-japanese"));
+ Metadata metadata = parse("/test-documents/testJPEG_EXIF.jpg");
+ assertEquals("2009-08-11T09:09:45",
metadata.get(TikaCoreProperties.CREATED));
+ assertEquals("2009-10-02T23:02:49",
metadata.get(TikaCoreProperties.MODIFIED));
+ assertEquals("2009-08-11T09:09:45",
metadata.get(TIFF.ORIGINAL_DATE));
+
+ metadata = parse("/test-documents/testJPEG_GEO_2.jpg");
+ assertEquals("2012-02-20T16:44:22Z",
metadata.get(Geographic.TIMESTAMP));
+ } finally {
+ Locale.setDefault(defaultLocale);
+ }
+ }
+
+ private static Metadata parse(String resource) throws Exception {
+ Metadata metadata = new Metadata();
+ try (InputStream is =
ImageMetadataExtractorLocaleTest.class.getResourceAsStream(resource);
+ TikaInputStream tis = TikaInputStream.get(is)) {
+ new JpegParser().parse(tis, new DefaultHandler(), metadata, new
ParseContext());
+ }
+ return metadata;
+ }
}
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorTest.java
index ab93eab488..37b94732a6 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorTest.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-image-module/src/test/java/org/apache/tika/parser/image/ImageMetadataExtractorTest.java
@@ -25,11 +25,8 @@ import java.nio.ByteBuffer;
import java.nio.charset.StandardCharsets;
import java.util.ArrayList;
import java.util.Arrays;
-import java.util.GregorianCalendar;
import java.util.Iterator;
import java.util.List;
-import java.util.Locale;
-import java.util.TimeZone;
import com.drew.lang.Rational;
import com.drew.metadata.Directory;
@@ -80,11 +77,7 @@ public class ImageMetadataExtractorTest {
public void testExifHandlerParseDate() throws MetadataException {
ExifSubIFDDirectory exif = Mockito.mock(ExifSubIFDDirectory.class);
Mockito.when(exif.containsTag(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn(true);
- GregorianCalendar calendar = new
GregorianCalendar(TimeZone.getTimeZone("UTC"), Locale.ROOT);
- calendar.setTimeInMillis(0);
- calendar.set(2000, 0, 1, 0, 0, 0);
- Mockito.when(exif.getDate(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL,
TimeZone.getTimeZone("GMT")))
- .thenReturn(calendar.getTime());
+
Mockito.when(exif.getObject(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn("2000:01:01
00:00:00");
Metadata metadata = new Metadata();
new ImageMetadataExtractor.ExifHandler().handle(exif, metadata);
@@ -118,11 +111,7 @@ public class ImageMetadataExtractorTest {
public void testExifHandlerParseDateFallback() throws MetadataException {
ExifIFD0Directory exif = Mockito.mock(ExifIFD0Directory.class);
Mockito.when(exif.containsTag(ExifIFD0Directory.TAG_DATETIME)).thenReturn(true);
- GregorianCalendar calendar = new
GregorianCalendar(TimeZone.getTimeZone("UTC"), Locale.ROOT);
- calendar.setTimeInMillis(0);
- calendar.set(1999, 0, 1, 0, 0, 0);
- Mockito.when(exif.getDate(ExifIFD0Directory.TAG_DATETIME,
TimeZone.getTimeZone("GMT")))
- .thenReturn(calendar.getTime());
+
Mockito.when(exif.getObject(ExifIFD0Directory.TAG_DATETIME)).thenReturn("1999:01:01
00:00:00");
Metadata metadata = new Metadata();
new ImageMetadataExtractor.ExifHandler().handle(exif, metadata);
@@ -134,11 +123,7 @@ public class ImageMetadataExtractorTest {
public void testExifHandlerParseDateOutOfBounds() throws MetadataException
{
ExifSubIFDDirectory exif = Mockito.mock(ExifSubIFDDirectory.class);
Mockito.when(exif.containsTag(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn(true);
- GregorianCalendar calendar = new
GregorianCalendar(TimeZone.getTimeZone("UTC"), Locale.ROOT);
- calendar.setTimeInMillis(0);
- calendar.set(4, 11, 31, 23, 0, 0);
- Mockito.when(exif.getDate(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL,
TimeZone.getTimeZone("GMT")))
- .thenReturn(calendar.getTime());
+
Mockito.when(exif.getObject(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn("0004:12:31
23:00:00");
Metadata metadata = new Metadata();
new ImageMetadataExtractor.ExifHandler().handle(exif, metadata);
assertNull(metadata.get(TikaCoreProperties.CREATED));
@@ -149,7 +134,7 @@ public class ImageMetadataExtractorTest {
public void testExifHandlerParseDateError() throws MetadataException {
ExifIFD0Directory exif = Mockito.mock(ExifIFD0Directory.class);
Mockito.when(exif.containsTag(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn(true);
- Mockito.when(exif.getDate(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL,
TimeZone.getTimeZone("GMT"))).thenReturn(null);
+
Mockito.when(exif.getObject(ExifSubIFDDirectory.TAG_DATETIME_ORIGINAL)).thenReturn("
: : : : ");
Metadata metadata = new Metadata();
new ImageMetadataExtractor.ExifHandler().handle(exif, metadata);
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
index f2d9240fb6..a5b9c05992 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-microsoft-module/src/test/java/org/apache/tika/parser/microsoft/JackcessParserTest.java
@@ -80,8 +80,8 @@ public class JackcessParserTest extends TikaTest {
}
}
- // TODO TIKA-4924: jackcess-encrypt uppercases cipher params in the
default locale
- @DisabledIfSystemProperty(named = "user.language", matches = "tr")
+ // TODO TIKA-4924: jackcess-encrypt uppercases cipher params in the
default locale (tr/az dotless i)
+ @DisabledIfSystemProperty(named = "tika.test.locale", matches =
"(tr|az)(-.*)?")
@Test
public void testPassword() throws Exception {
ParseContext c = new ParseContext();
diff --git
a/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/PerClientServerManager.java
b/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/PerClientServerManager.java
index 98a9b870f0..59390013f4 100644
---
a/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/PerClientServerManager.java
+++
b/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/PerClientServerManager.java
@@ -28,6 +28,7 @@ import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
+import java.util.Locale;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.TimeoutException;
@@ -670,6 +671,7 @@ public class PerClientServerManager implements
ServerManager {
boolean hasLog4j = false;
boolean hasActiveProcessorCount = false;
boolean hasErrorFile = false;
+ boolean hasLocale = false;
String origGCString = null;
String newGCLogString = null;
@@ -692,6 +694,9 @@ public class PerClientServerManager implements
ServerManager {
if (arg.startsWith("-XX:ErrorFile=")) {
hasErrorFile = true;
}
+ if (arg.startsWith("-Duser.language")) {
+ hasLocale = true;
+ }
if (arg.startsWith("-Xloggc:")) {
origGCString = arg;
newGCLogString = arg.replace("${pipesClientId}", "id-" +
clientId);
@@ -774,6 +779,10 @@ public class PerClientServerManager implements
ServerManager {
commandLine.add("-Dlog4j.configurationFile=classpath:pipes-fork-server-default-log4j2.xml");
}
commandLine.add("-DpipesClientId=" + clientId);
+ // the fork parses like the parent would; a fresh JVM would take the
OS locale instead
+ if (!hasLocale) {
+
commandLine.addAll(ProcessUtils.defaultLocaleJvmArgs(Locale.getDefault()));
+ }
commandLine.addAll(configArgs);
commandLine.add("-Djava.io.tmpdir=" + tmpDir.toAbsolutePath());
commandLine.add("org.apache.tika.pipes.core.server.PipesServer");
diff --git
a/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/SharedServerManager.java
b/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/SharedServerManager.java
index 8374d837b4..f39383cb69 100644
---
a/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/SharedServerManager.java
+++
b/tika-pipes/tika-pipes-core/src/main/java/org/apache/tika/pipes/core/SharedServerManager.java
@@ -28,6 +28,7 @@ import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
+import java.util.Locale;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.TimeoutException;
import java.util.concurrent.atomic.AtomicLong;
@@ -504,6 +505,7 @@ public class SharedServerManager implements ServerManager {
boolean hasExitOnOOM = false;
boolean hasLog4j = false;
boolean hasErrorFile = false;
+ boolean hasLocale = false;
for (String arg : configArgs) {
if (arg.startsWith("-Djava.awt.headless")) {
@@ -521,6 +523,9 @@ public class SharedServerManager implements ServerManager {
if (arg.startsWith("-XX:ErrorFile=")) {
hasErrorFile = true;
}
+ if (arg.startsWith("-Duser.language")) {
+ hasLocale = true;
+ }
}
// Direct native-crash dumps (hs_err_pid<N>.log) into tmpDir so
@@ -553,6 +558,10 @@ public class SharedServerManager implements ServerManager {
commandLine.add("-Dlog4j.configurationFile=classpath:pipes-fork-server-default-log4j2.xml");
}
commandLine.add("-DpipesClientId=shared");
+ // the fork parses like the parent would; a fresh JVM would take the
OS locale instead
+ if (!hasLocale) {
+
commandLine.addAll(ProcessUtils.defaultLocaleJvmArgs(Locale.getDefault()));
+ }
commandLine.addAll(configArgs);
commandLine.add("-Djava.io.tmpdir=" + tmpDir.toAbsolutePath());
commandLine.add("org.apache.tika.pipes.core.server.PipesServer");
diff --git
a/tika-pipes/tika-pipes-core/src/test/java/org/apache/tika/pipes/core/PerClientServerManagerLocaleTest.java
b/tika-pipes/tika-pipes-core/src/test/java/org/apache/tika/pipes/core/PerClientServerManagerLocaleTest.java
new file mode 100644
index 0000000000..57c99a3205
--- /dev/null
+++
b/tika-pipes/tika-pipes-core/src/test/java/org/apache/tika/pipes/core/PerClientServerManagerLocaleTest.java
@@ -0,0 +1,70 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.pipes.core;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+
+import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.Arrays;
+import java.util.List;
+import java.util.Locale;
+
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.io.TempDir;
+import org.junit.jupiter.api.parallel.Isolated;
+
+/** The fork gets the parent's default locale unless forkedJvmArgs set one. */
+// sets the JVM-wide default locale
+@Isolated
+public class PerClientServerManagerLocaleTest {
+
+ @TempDir
+ Path tmp;
+
+ private List<String> userArgs(String... forkedJvmArgs) throws Exception {
+ PipesConfig pipesConfig = new PipesConfig();
+ pipesConfig.setForkedJvmArgs(new
ArrayList<>(Arrays.asList(forkedJvmArgs)));
+ PerClientServerManager manager =
+ new PerClientServerManager(pipesConfig, new byte[0], 0);
+ return Arrays.stream(manager.getCommandline(tmp))
+ .filter(a -> a.startsWith("-Duser.")).toList();
+ }
+
+ @Test
+ public void testParentLocalePropagates() throws Exception {
+ Locale defaultLocale = Locale.getDefault();
+ try {
+
Locale.setDefault(Locale.forLanguageTag("th-TH-u-nu-thai-x-lvariant-TH"));
+ assertEquals(List.of("-Duser.language=th", "-Duser.country=TH",
"-Duser.variant=TH",
+ "-Duser.extensions=u-nu-thai"), userArgs());
+ Locale.setDefault(Locale.forLanguageTag("sr-Latn-RS"));
+ assertEquals(List.of("-Duser.language=sr", "-Duser.script=Latn",
"-Duser.country=RS"),
+ userArgs());
+ Locale.setDefault(Locale.ROOT);
+ assertEquals(List.of(), userArgs());
+ } finally {
+ Locale.setDefault(defaultLocale);
+ }
+ }
+
+ @Test
+ public void testConfiguredLocaleWins() throws Exception {
+ assertEquals(List.of("-Duser.language=tr", "-Duser.country=TR"),
+ userArgs("-Duser.language=tr", "-Duser.country=TR"));
+ }
+}
diff --git a/tika-test-support/pom.xml b/tika-test-support/pom.xml
new file mode 100644
index 0000000000..88946b628f
--- /dev/null
+++ b/tika-test-support/pom.xml
@@ -0,0 +1,56 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<!--
+ Licensed to the Apache Software Foundation (ASF) under one
+ or more contributor license agreements. See the NOTICE file
+ distributed with this work for additional information
+ regarding copyright ownership. The ASF licenses this file
+ to you under the Apache License, Version 2.0 (the
+ "License"); you may not use this file except in compliance
+ with the License. You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+ Unless required by applicable law or agreed to in writing,
+ software distributed under the License is distributed on an
+ "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ KIND, either express or implied. See the License for the
+ specific language governing permissions and limitations
+ under the License.
+-->
+<project xmlns="http://maven.apache.org/POM/4.0.0"
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0
https://maven.apache.org/xsd/maven-4.0.0.xsd">
+ <modelVersion>4.0.0</modelVersion>
+ <parent>
+ <groupId>org.apache.tika</groupId>
+ <artifactId>tika-parent</artifactId>
+ <version>${revision}</version>
+ <relativePath>../tika-parent/pom.xml</relativePath>
+ </parent>
+ <artifactId>tika-test-support</artifactId>
+ <name>Apache Tika test support</name>
+ <description>
+ JUnit Platform listeners on every module's test classpath: a random default
+ locale per test JVM, reported on failure.
+ </description>
+ <url>https://tika.apache.org</url>
+ <dependencies>
+ <dependency>
+ <groupId>org.junit.platform</groupId>
+ <artifactId>junit-platform-launcher</artifactId>
+ </dependency>
+ </dependencies>
+ <build>
+ <plugins>
+ <plugin>
+ <groupId>org.apache.maven.plugins</groupId>
+ <artifactId>maven-jar-plugin</artifactId>
+ <configuration>
+ <archive>
+ <manifestEntries>
+
<Automatic-Module-Name>org.apache.tika.test.support</Automatic-Module-Name>
+ </manifestEntries>
+ </archive>
+ </configuration>
+ </plugin>
+ </plugins>
+ </build>
+</project>
diff --git
a/tika-test-support/src/main/java/org/apache/tika/test/RandomLocaleListener.java
b/tika-test-support/src/main/java/org/apache/tika/test/RandomLocaleListener.java
new file mode 100644
index 0000000000..b633064735
--- /dev/null
+++
b/tika-test-support/src/main/java/org/apache/tika/test/RandomLocaleListener.java
@@ -0,0 +1,122 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.test;
+
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
+import java.util.concurrent.ThreadLocalRandom;
+
+import org.junit.platform.engine.TestExecutionResult;
+import org.junit.platform.launcher.LauncherSession;
+import org.junit.platform.launcher.LauncherSessionListener;
+import org.junit.platform.launcher.TestExecutionListener;
+import org.junit.platform.launcher.TestIdentifier;
+
+/**
+ * With {@code -Dtika.test.locale=random}, runs the test JVM under a random
default locale, as
+ * Lucene's test framework does, so locale-sensitive code paths are exercised
across the
+ * JDK's whole locale set over time rather than in one pinned locale. CI
passes that flag;
+ * a plain build keeps the JVM's own locale, so release and developer builds
are
+ * deterministic. Registered through {@code META-INF/services}; the launcher
picks it up
+ * from the test classpath.
+ * <p>
+ * {@code -Dtika.test.locale=<language tag>} pins the locale (as printed by a
failed run).
+ * The effective language tag is published under the same property, and every
failure
+ * carries it as a suppressed exception so a surefire report shows how to
reproduce.
+ */
+public class RandomLocaleListener implements LauncherSessionListener,
TestExecutionListener {
+
+ public static final String PROPERTY = "tika.test.locale";
+
+ /**
+ * Half of the random picks come from here: locales whose casing, digit,
calendar or
+ * shaping rules differ from en-US in ways code often overlooks, and which
uniform
+ * sampling would each reach only once in several hundred runs.
+ */
+ static final List<Locale> PRIORITY_LOCALES = List.of(
+ Locale.forLanguageTag("tr-TR"),
+ Locale.forLanguageTag("az-Latn-AZ"),
+ Locale.forLanguageTag("th-TH-u-nu-thai-x-lvariant-TH"),
+ Locale.forLanguageTag("ja-JP-u-ca-japanese-x-lvariant-JP"),
+ Locale.forLanguageTag("ar-EG"),
+ Locale.forLanguageTag("de-DE"));
+
+ private static volatile boolean reported;
+
+ @Override
+ public void launcherSessionOpened(LauncherSession session) {
+ Locale locale = choose(System.getProperty(PROPERTY));
+ Locale.setDefault(locale);
+ System.setProperty(PROPERTY, locale.toLanguageTag());
+ System.out.println("[tika] default locale " + locale.toLanguageTag() +
+ " (reproduce with -D" + PROPERTY + "=" +
locale.toLanguageTag() + ")");
+ }
+
+ static Locale choose(String setting) {
+ if (setting == null || setting.isEmpty() || "system".equals(setting)) {
+ return Locale.getDefault();
+ }
+ if ("random".equals(setting)) {
+ ThreadLocalRandom random = ThreadLocalRandom.current();
+ List<Locale> candidates = random.nextBoolean() ? PRIORITY_LOCALES
: candidates();
+ return candidates.get(random.nextInt(candidates.size()));
+ }
+ Locale locale = Locale.forLanguageTag(setting);
+ if (locale.getLanguage().isEmpty()) {
+ throw new IllegalArgumentException("not a language tag: -D" +
PROPERTY + "=" + setting);
+ }
+ return locale;
+ }
+
+ /** Available locales whose language tag reproduces them; no_NO_NY, for
one, does not. */
+ static List<Locale> candidates() {
+ List<Locale> candidates = new ArrayList<>();
+ for (Locale locale : Locale.getAvailableLocales()) {
+ if (!locale.getLanguage().isEmpty() &&
+
locale.equals(Locale.forLanguageTag(locale.toLanguageTag()))) {
+ candidates.add(locale);
+ }
+ }
+ return candidates;
+ }
+
+ @Override
+ public void executionFinished(TestIdentifier identifier,
TestExecutionResult result) {
+ if (result.getStatus() != TestExecutionResult.Status.FAILED) {
+ return;
+ }
+ String tag = System.getProperty(PROPERTY);
+ if (tag == null) {
+ return;
+ }
+ result.getThrowable().ifPresent(t -> t.addSuppressed(new
LocaleNote(tag)));
+ if (!reported) {
+ reported = true;
+ System.err.println("[tika] failure under default locale " + tag +
+ "; reproduce with -D" + PROPERTY + "=" + tag);
+ }
+ }
+
+ /** Attached to a failure so the reproduction hint survives into the
surefire report. */
+ public static final class LocaleNote extends Exception {
+ LocaleNote(String tag) {
+ super("default locale was " + tag + "; reproduce with -D" +
PROPERTY + "=" + tag,
+ null, false, false);
+ }
+ }
+}
diff --git
a/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.LauncherSessionListener
b/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.LauncherSessionListener
new file mode 100644
index 0000000000..f4c1a97825
--- /dev/null
+++
b/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.LauncherSessionListener
@@ -0,0 +1,17 @@
+#
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements. See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+org.apache.tika.test.RandomLocaleListener
diff --git
a/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.TestExecutionListener
b/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.TestExecutionListener
new file mode 100644
index 0000000000..f4c1a97825
--- /dev/null
+++
b/tika-test-support/src/main/resources/META-INF/services/org.junit.platform.launcher.TestExecutionListener
@@ -0,0 +1,17 @@
+#
+# Licensed to the Apache Software Foundation (ASF) under one or more
+# contributor license agreements. See the NOTICE file distributed with
+# this work for additional information regarding copyright ownership.
+# The ASF licenses this file to You under the Apache License, Version 2.0
+# (the "License"); you may not use this file except in compliance with
+# the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+org.apache.tika.test.RandomLocaleListener