This is an automated email from the ASF dual-hosted git repository.

tballison pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/tika.git


The following commit(s) were added to refs/heads/main by this push:
     new 758a57904f TIKA-4774 -- defer to structural a bit more (#2976)
758a57904f is described below

commit 758a57904fe4cbb189e2602a97e55c631ea8dcc1
Author: Tim Allison <[email protected]>
AuthorDate: Wed Jul 29 15:53:05 2026 -0400

    TIKA-4774 -- defer to structural a bit more (#2976)
---
 .../ml/chardetect/StructuralEncodingRules.java     | 37 +++++++++
 .../ml/junkdetect/JunkFilterEncodingDetector.java  | 46 +++++++++++
 .../junkdetect/JunkFilterEncodingDetectorTest.java | 90 ++++++++++++++++++++++
 3 files changed, 173 insertions(+)

diff --git 
a/tika-encoding-detectors/tika-encoding-detector-mojibuster/src/main/java/org/apache/tika/ml/chardetect/StructuralEncodingRules.java
 
b/tika-encoding-detectors/tika-encoding-detector-mojibuster/src/main/java/org/apache/tika/ml/chardetect/StructuralEncodingRules.java
index 2c46a8db3a..2302f110d4 100644
--- 
a/tika-encoding-detectors/tika-encoding-detector-mojibuster/src/main/java/org/apache/tika/ml/chardetect/StructuralEncodingRules.java
+++ 
b/tika-encoding-detectors/tika-encoding-detector-mojibuster/src/main/java/org/apache/tika/ml/chardetect/StructuralEncodingRules.java
@@ -749,6 +749,43 @@ public final class StructuralEncodingRules {
         return Utf8Result.AMBIGUOUS;
     }
 
+    /**
+     * True if the sample has a COMPLETE 3-/4-byte UTF-8 sequence (lead
+     * {@code 0xE0–0xF4}, continuations {@code 0x80–0xBF}).  A legacy 2-byte 
CJK
+     * pair can fake a 2-byte UTF-8 char but never a wide one — the 
discriminator
+     * between real UTF-8 and a CJK coincidence.  Tests width only; caller must
+     * already know the sample is valid UTF-8 (see {@link #checkUtf8}).
+     */
+    public static boolean hasWideUtf8Sequence(byte[] bytes) {
+        return hasWideUtf8Sequence(bytes, 0, bytes.length);
+    }
+
+    public static boolean hasWideUtf8Sequence(byte[] bytes, int offset, int 
length) {
+        int end = offset + length;
+        for (int i = offset; i < end; i++) {
+            int b = bytes[i] & 0xFF;
+            if (b < 0xE0 || b > 0xF4) {
+                continue;
+            }
+            int seqLen = (b >= 0xF0) ? 4 : 3;
+            if (i + seqLen > end) {
+                continue; // truncated lead at probe end — not proof
+            }
+            boolean complete = true;
+            for (int k = 1; k < seqLen; k++) {
+                int cb = bytes[i + k] & 0xFF;
+                if (cb < 0x80 || cb > 0xBF) {
+                    complete = false;
+                    break;
+                }
+            }
+            if (complete) {
+                return true;
+            }
+        }
+        return false;
+    }
+
     /**
      * Counts the number of malformed UTF-8 <em>sequences</em> in the sample —
      * one event per bad lead, orphaned continuation, overlong, surrogate, or
diff --git 
a/tika-ml/tika-ml-junkdetect/src/main/java/org/apache/tika/ml/junkdetect/JunkFilterEncodingDetector.java
 
b/tika-ml/tika-ml-junkdetect/src/main/java/org/apache/tika/ml/junkdetect/JunkFilterEncodingDetector.java
index 76af07b373..46f100e63d 100644
--- 
a/tika-ml/tika-ml-junkdetect/src/main/java/org/apache/tika/ml/junkdetect/JunkFilterEncodingDetector.java
+++ 
b/tika-ml/tika-ml-junkdetect/src/main/java/org/apache/tika/ml/junkdetect/JunkFilterEncodingDetector.java
@@ -18,6 +18,7 @@ package org.apache.tika.ml.junkdetect;
 
 import java.io.IOException;
 import java.nio.charset.Charset;
+import java.nio.charset.StandardCharsets;
 import java.util.ArrayList;
 import java.util.Arrays;
 import java.util.Collections;
@@ -42,6 +43,7 @@ import org.apache.tika.detect.MetaEncodingDetector;
 import org.apache.tika.io.TikaInputStream;
 import org.apache.tika.metadata.Metadata;
 import org.apache.tika.ml.chardetect.AdaptiveProbe;
+import org.apache.tika.ml.chardetect.StructuralEncodingRules;
 import org.apache.tika.parser.ParseContext;
 import org.apache.tika.quality.TextQualityDetector;
 
@@ -203,6 +205,20 @@ public class JunkFilterEncodingDetector implements 
MetaEncodingDetector {
         // decoded as the author intended.
         Charset declared = pickDeclarativeWithEquivalentDecode(context, 
candidates);
         if (declared != null) {
+            // A STRUCTURAL proof (valid multi-byte UTF-8, ISO-2022, BOM) 
outranks a
+            // DECLARATIVE self-claim on conflict.  The equivalent-decode 
shortcut
+            // compares tag-stripped text, so a doc whose only high bytes live 
in
+            // markup (a <meta> attribute, an <o:..> tag) ties on visible text 
and
+            // would honour a declaration the bytes contradict.  Defer to the 
proof
+            // (CJK-declaration guard in #overridesDeclared).
+            Charset structural = conflictingStructural(context, candidates, 
declared);
+            if (structural != null && overridesDeclared(structural, declared, 
bytes)) {
+                context.setArbitrationInfo("junk-filter-prefer-structural");
+                LOG.trace("junk-filter -> {} (structural proof overrides 
declared {})",
+                        structural.name(), declared.name());
+                return List.of(new EncodingResult(structural,
+                        context.getTopConfidenceFor(structural)));
+            }
             float conf = context.getTopConfidenceFor(declared);
             context.setArbitrationInfo("junk-filter-prefer-declarative");
             LOG.trace("junk-filter -> {} (declarative with equivalent decode)",
@@ -581,6 +597,36 @@ public class JunkFilterEncodingDetector implements 
MetaEncodingDetector {
         return null;
     }
 
+    /** A decoded STRUCTURAL candidate other than the declared charset — 
evidence
+     *  that can outrank the declaration; {@code null} if none. */
+    private static Charset conflictingStructural(EncodingDetectorContext 
context,
+            Map<Charset, String> candidates, Charset declared) {
+        for (EncodingDetectorContext.Result r : context.getResults()) {
+            for (EncodingResult er : r.getEncodingResults()) {
+                if (er.getResultType() != 
EncodingResult.ResultType.STRUCTURAL) {
+                    continue;
+                }
+                Charset cs = er.getCharset();
+                if (!cs.equals(declared) && candidates.containsKey(cs)) {
+                    return cs;
+                }
+            }
+        }
+        return null;
+    }
+
+    /** Whether a conflicting STRUCTURAL proof overrides the declaration.  
UTF-8 vs
+     *  a variable-length CJK declaration is ambiguous on a short probe (a 
2-byte CJK
+     *  pair can fake a 2-byte UTF-8 char, never a wide one) → require a wide
+     *  sequence; all other proofs are unambiguous. */
+    private static boolean overridesDeclared(Charset structural, Charset 
declared,
+            byte[] bytes) {
+        if (StandardCharsets.UTF_8.equals(structural) && 
isCjkCharset(declared.name())) {
+            return StructuralEncodingRules.hasWideUtf8Sequence(bytes);
+        }
+        return true;
+    }
+
     private static boolean allDecodingsIdentical(Map<Charset, String> 
candidates) {
         String first = null;
         for (String s : candidates.values()) {
diff --git 
a/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/JunkFilterEncodingDetectorTest.java
 
b/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/JunkFilterEncodingDetectorTest.java
index a0904289b8..f1ae65a856 100644
--- 
a/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/JunkFilterEncodingDetectorTest.java
+++ 
b/tika-ml/tika-ml-junkdetect/src/test/java/org/apache/tika/ml/junkdetect/JunkFilterEncodingDetectorTest.java
@@ -231,6 +231,96 @@ public class JunkFilterEncodingDetectorTest {
         }
     }
 
+    /** {@code <a x="..">plain body</a>} with {@code hi} as the attribute's 
high
+     *  bytes — all non-ASCII lives in markup, so every candidate's 
tag-stripped
+     *  text ties and the equivalent-decode shortcut engages. */
+    private static byte[] markupWithHighBytes(byte[] hi) {
+        byte[] pre = "<a x=\"".getBytes(StandardCharsets.US_ASCII);
+        byte[] post = "\">plain ascii body text 
here</a>".getBytes(StandardCharsets.US_ASCII);
+        byte[] out = new byte[pre.length + hi.length + post.length];
+        System.arraycopy(pre, 0, out, 0, pre.length);
+        System.arraycopy(hi, 0, out, pre.length, hi.length);
+        System.arraycopy(post, 0, out, pre.length + hi.length, post.length);
+        return out;
+    }
+
+    @Test
+    public void structuralUtf8OverridesDeclaredLatin() throws Exception {
+        // Real-bug shape (TIKA test.html): declared x-MacRoman but bytes are 
UTF-8
+        // (é = C3 A9) with the only high bytes in markup, so stripped text 
ties.
+        // A STRUCTURAL proof must beat the contradicted declaration.
+        Charset utf8 = StandardCharsets.UTF_8;
+        Charset macRoman = Charset.forName("x-MacRoman");
+        byte[] bytes = markupWithHighBytes(new byte[] {(byte) 0xC3, (byte) 
0xA9});
+
+        ParseContext pc = contextWith(
+                new EncodingResult(macRoman, 0.9f, "x-MacRoman",
+                        EncodingResult.ResultType.DECLARATIVE),
+                new EncodingResult(utf8, 0.95f, "UTF-8",
+                        EncodingResult.ResultType.STRUCTURAL));
+
+        JunkFilterEncodingDetector detector =
+                new JunkFilterEncodingDetector(new PreferenceStub("UTF-8"));
+        try (TikaInputStream tis = TikaInputStream.get(bytes)) {
+            List<EncodingResult> out = detector.detect(tis, new Metadata(), 
pc);
+            assertEquals(1, out.size());
+            assertEquals(utf8, out.get(0).getCharset());
+            assertEquals("junk-filter-prefer-structural",
+                    
pc.get(EncodingDetectorContext.class).getArbitrationInfo());
+        }
+    }
+
+    @Test
+    public void structuralUtf8DefersToDeclaredCjkWithoutWideSequence() throws 
Exception {
+        // Declared Shift_JIS + STRUCTURAL UTF-8, but the only UTF-8 evidence 
is a
+        // 2-byte sequence a legacy CJK pair can fake — keep the declaration.
+        Charset utf8 = StandardCharsets.UTF_8;
+        Charset sjis = Charset.forName("Shift_JIS");
+        byte[] bytes = markupWithHighBytes(new byte[] {(byte) 0xC3, (byte) 
0xA9});
+
+        ParseContext pc = contextWith(
+                new EncodingResult(sjis, 0.9f, "Shift_JIS",
+                        EncodingResult.ResultType.DECLARATIVE),
+                new EncodingResult(utf8, 0.95f, "UTF-8",
+                        EncodingResult.ResultType.STRUCTURAL));
+
+        JunkFilterEncodingDetector detector =
+                new JunkFilterEncodingDetector(new PreferenceStub("UTF-8"));
+        try (TikaInputStream tis = TikaInputStream.get(bytes)) {
+            List<EncodingResult> out = detector.detect(tis, new Metadata(), 
pc);
+            assertEquals(1, out.size());
+            assertEquals(sjis, out.get(0).getCharset());
+            assertEquals("junk-filter-prefer-declarative",
+                    
pc.get(EncodingDetectorContext.class).getArbitrationInfo());
+        }
+    }
+
+    @Test
+    public void structuralUtf8OverridesDeclaredCjkWithWideSequence() throws 
Exception {
+        // Declared Shift_JIS + STRUCTURAL UTF-8 with a 3-byte sequence (日 =
+        // E6 97 A5): unfakeable by a legacy 2-byte CJK pair, so the proof 
wins.
+        Charset utf8 = StandardCharsets.UTF_8;
+        Charset sjis = Charset.forName("Shift_JIS");
+        byte[] bytes = markupWithHighBytes(
+                new byte[] {(byte) 0xE6, (byte) 0x97, (byte) 0xA5});
+
+        ParseContext pc = contextWith(
+                new EncodingResult(sjis, 0.9f, "Shift_JIS",
+                        EncodingResult.ResultType.DECLARATIVE),
+                new EncodingResult(utf8, 0.95f, "UTF-8",
+                        EncodingResult.ResultType.STRUCTURAL));
+
+        JunkFilterEncodingDetector detector =
+                new JunkFilterEncodingDetector(new PreferenceStub("UTF-8"));
+        try (TikaInputStream tis = TikaInputStream.get(bytes)) {
+            List<EncodingResult> out = detector.detect(tis, new Metadata(), 
pc);
+            assertEquals(1, out.size());
+            assertEquals(utf8, out.get(0).getCharset());
+            assertEquals("junk-filter-prefer-structural",
+                    
pc.get(EncodingDetectorContext.class).getArbitrationInfo());
+        }
+    }
+
     // NOTE: a full default-constructor integration test (which would load
     // the bundled JunkDetector via ServiceLoader) is not included here
     // because JunkDetector currently exposes only static factory methods

Reply via email to