This is an automated email from the ASF dual-hosted git repository.

tballison pushed a commit to branch branch_3x
in repository https://gitbox.apache.org/repos/asf/tika.git


The following commit(s) were added to refs/heads/branch_3x by this push:
     new 0f8067b0f2 TIKA-4796: compile MagicDetector regex once at construction 
(#2972)
0f8067b0f2 is described below

commit 0f8067b0f2aad210bed54cb33340e02ae27702b8
Author: Grant Ingersoll <[email protected]>
AuthorDate: Tue Jul 28 17:31:17 2026 -0400

    TIKA-4796: compile MagicDetector regex once at construction (#2972)
    
    Backport of 65a2f3ab6 (#2968) from main to branch_3x.
    
    MagicDetector recompiled its Pattern on every regex detect() even though
    the pattern bytes and the case-insensitivity flag are both fixed at
    construction time. Compile once in the constructor and reuse the
    Pattern; Pattern is immutable and its Matcher is still created per
    call, so the regex path stays thread-safe.
    
    The 3.x code compiles the Pattern inline in detect() rather than in the
    matchesBuffer/matches(byte[]) helpers that main has, so this is a
    hand-ported diff rather than a cherry-pick. The main-branch test that
    covered the regex branch of matches(byte[]) has no counterpart here --
    3.x already exercises the regex detect() path via testDetectRegExPDF,
    testDetectRegExGreedy and testDetectRegExOptions -- so only the
    repeated-call and concurrent-use guards over a shared detector instance
    are carried over.
    
    
    Claude-Session: https://claude.ai/code/session_01Pp6TVDKF7ztw7SS7hwwuFd
    
    Co-authored-by: Claude Opus 5 <[email protected]>
---
 CHANGES.txt                                        |  5 ++
 .../java/org/apache/tika/detect/MagicDetector.java | 34 ++++++++---
 .../org/apache/tika/detect/MagicDetectorTest.java  | 65 ++++++++++++++++++++++
 3 files changed, 96 insertions(+), 8 deletions(-)

diff --git a/CHANGES.txt b/CHANGES.txt
index caa8ce95dc..84f1378c32 100644
--- a/CHANGES.txt
+++ b/CHANGES.txt
@@ -1,3 +1,8 @@
+Release 3.3.3 - ???
+
+  * MagicDetector now compiles its regular expression once, in the
+    constructor, instead of recompiling it on every match (TIKA-4796).
+
 Release 3.3.2 - 7/16/2026
 
   * Dependency upgrades (TIKA-4738).
diff --git a/tika-core/src/main/java/org/apache/tika/detect/MagicDetector.java 
b/tika-core/src/main/java/org/apache/tika/detect/MagicDetector.java
index 5e00779be2..25d02844e8 100644
--- a/tika-core/src/main/java/org/apache/tika/detect/MagicDetector.java
+++ b/tika-core/src/main/java/org/apache/tika/detect/MagicDetector.java
@@ -39,6 +39,9 @@ import org.apache.tika.mime.MediaType;
  * Because this works on bytes, not characters, by default any string
  * matching is done as ISO_8859_1. To use an explicit different
  * encoding, supply a type other than "string" / "stringignorecase"
+ * <p>
+ * Instances of this class are immutable and safe for use by multiple
+ * concurrent threads.
  *
  * @since Apache Tika 0.3
  */
@@ -91,6 +94,12 @@ public class MagicDetector implements Detector {
      * starts at this offset.
      */
     private final int offsetRangeEnd;
+    /**
+     * The compiled form of {@link #pattern} when {@link #isRegex} is true,
+     * <code>null</code> otherwise. Compiled once here rather than per match,
+     * as every input to it is fixed at construction time.
+     */
+    private final Pattern compiledPattern;
 
     /**
      * Creates a detector for input documents that have the exact given byte
@@ -137,6 +146,15 @@ public class MagicDetector implements Detector {
     /**
      * Creates a detector for input documents that meet the specified
      * magic match.
+     * <p>
+     * When <code>isRegex</code> is true the pattern is compiled here rather
+     * than on each match, so a malformed pattern is reported by this
+     * constructor instead of by the first call to
+     * {@link #detect(InputStream, Metadata)}.
+     *
+     * @throws java.util.regex.PatternSyntaxException if <code>isRegex</code>
+     *         is true and <code>pattern</code> is not a valid regular
+     *         expression
      */
     public MagicDetector(MediaType type, byte[] pattern, byte[] mask, boolean 
isRegex,
                          boolean isStringIgnoreCase, int offsetRangeBegin, int 
offsetRangeEnd) {
@@ -180,6 +198,13 @@ public class MagicDetector implements Detector {
             }
         }
 
+        if (this.isRegex) {
+            int flags = this.isStringIgnoreCase ? Pattern.CASE_INSENSITIVE : 0;
+            this.compiledPattern = Pattern.compile(new String(this.pattern, 
UTF_8), flags);
+        } else {
+            this.compiledPattern = null;
+        }
+
         this.offsetRangeBegin = offsetRangeBegin;
         this.offsetRangeEnd = offsetRangeEnd;
     }
@@ -374,16 +399,9 @@ public class MagicDetector implements Detector {
             }
 
             if (this.isRegex) {
-                int flags = 0;
-                if (this.isStringIgnoreCase) {
-                    flags = Pattern.CASE_INSENSITIVE;
-                }
-
-                Pattern p = Pattern.compile(new String(this.pattern, UTF_8), 
flags);
-
                 ByteBuffer bb = ByteBuffer.wrap(buffer);
                 CharBuffer result = ISO_8859_1.decode(bb);
-                Matcher m = p.matcher(result);
+                Matcher m = compiledPattern.matcher(result);
 
                 boolean match = false;
                 // Loop until we've covered the entire offset range
diff --git 
a/tika-core/src/test/java/org/apache/tika/detect/MagicDetectorTest.java 
b/tika-core/src/test/java/org/apache/tika/detect/MagicDetectorTest.java
index 3a86a53b36..6be5fce590 100644
--- a/tika-core/src/test/java/org/apache/tika/detect/MagicDetectorTest.java
+++ b/tika-core/src/test/java/org/apache/tika/detect/MagicDetectorTest.java
@@ -26,6 +26,12 @@ import java.io.BufferedInputStream;
 import java.io.ByteArrayInputStream;
 import java.io.IOException;
 import java.io.InputStream;
+import java.util.ArrayList;
+import java.util.List;
+import java.util.concurrent.ExecutorService;
+import java.util.concurrent.Executors;
+import java.util.concurrent.Future;
+import java.util.concurrent.TimeUnit;
 
 import org.apache.commons.io.IOUtils;
 import org.junit.jupiter.api.Test;
@@ -158,6 +164,65 @@ public class MagicDetectorTest {
         assertDetect(detector, MediaType.OCTET_STREAM, data2);
     }
 
+    /**
+     * A MagicDetector is built once and reused for the life of the process, so
+     * repeated calls must be independent of each other. Guards the compiled
+     * Pattern against per-call state leaking in.
+     */
+    @Test
+    public void testRegExDetectorRepeatedCallsStable() throws Exception {
+        MediaType html = new MediaType("text", "html");
+        String pattern = "(?s)\\A.{0,1024}\\x3c\\!(?:DOCTYPE|doctype) 
(?:HTML|html) ";
+        Detector detector =
+                new MagicDetector(html, pattern.getBytes(US_ASCII), null, 
true, 0, 0);
+
+        String match = "<!DOCTYPE HTML PUBLIC \"-//W3C//DTD HTML 4.01//EN\">";
+        String noMatch = "<html><head><title>plain</title></head>";
+
+        for (int i = 0; i < 100; i++) {
+            assertDetect(detector, html, match);
+            assertDetect(detector, MediaType.OCTET_STREAM, noMatch);
+        }
+    }
+
+    /**
+     * MimeTypes shares one MagicDetector instance per magic clause across 
every
+     * caller, so the regex path has to be safe to use concurrently.
+     */
+    @Test
+    public void testRegExDetectorConcurrent() throws Exception {
+        MediaType pdf = new MediaType("application", "pdf");
+        Detector detector =
+                new MagicDetector(pdf, 
"(?s)\\A.{0,144}%PDF-".getBytes(US_ASCII), null, true, 0, 0);
+
+        byte[] match = "%PDF-1.4\nsome trailing content".getBytes(US_ASCII);
+        byte[] noMatch = "not a pdf at all".getBytes(US_ASCII);
+
+        int threads = 8;
+        int iterations = 200;
+        ExecutorService executor = Executors.newFixedThreadPool(threads);
+        try {
+            List<Future<?>> futures = new ArrayList<>();
+            for (int t = 0; t < threads; t++) {
+                futures.add(executor.submit(() -> {
+                    for (int i = 0; i < iterations; i++) {
+                        assertEquals(pdf, detector.detect(new 
ByteArrayInputStream(match),
+                                new Metadata()));
+                        assertEquals(MediaType.OCTET_STREAM,
+                                detector.detect(new 
ByteArrayInputStream(noMatch), new Metadata()));
+                    }
+                    return null;
+                }));
+            }
+            for (Future<?> future : futures) {
+                // an assertion failure on a worker surfaces here as an 
ExecutionException
+                future.get(60, TimeUnit.SECONDS);
+            }
+        } finally {
+            executor.shutdownNow();
+        }
+    }
+
     @Test
     public void testDetectStreamReadProblems() throws Exception {
         byte[] data = 
"abcdefghijklmnopqrstuvwxyz0123456789".getBytes(US_ASCII);

Reply via email to