This is an automated email from the ASF dual-hosted git repository.
tballison pushed a commit to branch branch_3x
in repository https://gitbox.apache.org/repos/asf/tika.git
The following commit(s) were added to refs/heads/branch_3x by this push:
new 0f8067b0f2 TIKA-4796: compile MagicDetector regex once at construction
(#2972)
0f8067b0f2 is described below
commit 0f8067b0f2aad210bed54cb33340e02ae27702b8
Author: Grant Ingersoll <[email protected]>
AuthorDate: Tue Jul 28 17:31:17 2026 -0400
TIKA-4796: compile MagicDetector regex once at construction (#2972)
Backport of 65a2f3ab6 (#2968) from main to branch_3x.
MagicDetector recompiled its Pattern on every regex detect() even though
the pattern bytes and the case-insensitivity flag are both fixed at
construction time. Compile once in the constructor and reuse the
Pattern; Pattern is immutable and its Matcher is still created per
call, so the regex path stays thread-safe.
The 3.x code compiles the Pattern inline in detect() rather than in the
matchesBuffer/matches(byte[]) helpers that main has, so this is a
hand-ported diff rather than a cherry-pick. The main-branch test that
covered the regex branch of matches(byte[]) has no counterpart here --
3.x already exercises the regex detect() path via testDetectRegExPDF,
testDetectRegExGreedy and testDetectRegExOptions -- so only the
repeated-call and concurrent-use guards over a shared detector instance
are carried over.
Claude-Session: https://claude.ai/code/session_01Pp6TVDKF7ztw7SS7hwwuFd
Co-authored-by: Claude Opus 5 <[email protected]>
---
CHANGES.txt | 5 ++
.../java/org/apache/tika/detect/MagicDetector.java | 34 ++++++++---
.../org/apache/tika/detect/MagicDetectorTest.java | 65 ++++++++++++++++++++++
3 files changed, 96 insertions(+), 8 deletions(-)
diff --git a/CHANGES.txt b/CHANGES.txt
index caa8ce95dc..84f1378c32 100644
--- a/CHANGES.txt
+++ b/CHANGES.txt
@@ -1,3 +1,8 @@
+Release 3.3.3 - ???
+
+ * MagicDetector now compiles its regular expression once, in the
+ constructor, instead of recompiling it on every match (TIKA-4796).
+
Release 3.3.2 - 7/16/2026
* Dependency upgrades (TIKA-4738).
diff --git a/tika-core/src/main/java/org/apache/tika/detect/MagicDetector.java
b/tika-core/src/main/java/org/apache/tika/detect/MagicDetector.java
index 5e00779be2..25d02844e8 100644
--- a/tika-core/src/main/java/org/apache/tika/detect/MagicDetector.java
+++ b/tika-core/src/main/java/org/apache/tika/detect/MagicDetector.java
@@ -39,6 +39,9 @@ import org.apache.tika.mime.MediaType;
* Because this works on bytes, not characters, by default any string
* matching is done as ISO_8859_1. To use an explicit different
* encoding, supply a type other than "string" / "stringignorecase"
+ * <p>
+ * Instances of this class are immutable and safe for use by multiple
+ * concurrent threads.
*
* @since Apache Tika 0.3
*/
@@ -91,6 +94,12 @@ public class MagicDetector implements Detector {
* starts at this offset.
*/
private final int offsetRangeEnd;
+ /**
+ * The compiled form of {@link #pattern} when {@link #isRegex} is true,
+ * <code>null</code> otherwise. Compiled once here rather than per match,
+ * as every input to it is fixed at construction time.
+ */
+ private final Pattern compiledPattern;
/**
* Creates a detector for input documents that have the exact given byte
@@ -137,6 +146,15 @@ public class MagicDetector implements Detector {
/**
* Creates a detector for input documents that meet the specified
* magic match.
+ * <p>
+ * When <code>isRegex</code> is true the pattern is compiled here rather
+ * than on each match, so a malformed pattern is reported by this
+ * constructor instead of by the first call to
+ * {@link #detect(InputStream, Metadata)}.
+ *
+ * @throws java.util.regex.PatternSyntaxException if <code>isRegex</code>
+ * is true and <code>pattern</code> is not a valid regular
+ * expression
*/
public MagicDetector(MediaType type, byte[] pattern, byte[] mask, boolean
isRegex,
boolean isStringIgnoreCase, int offsetRangeBegin, int
offsetRangeEnd) {
@@ -180,6 +198,13 @@ public class MagicDetector implements Detector {
}
}
+ if (this.isRegex) {
+ int flags = this.isStringIgnoreCase ? Pattern.CASE_INSENSITIVE : 0;
+ this.compiledPattern = Pattern.compile(new String(this.pattern,
UTF_8), flags);
+ } else {
+ this.compiledPattern = null;
+ }
+
this.offsetRangeBegin = offsetRangeBegin;
this.offsetRangeEnd = offsetRangeEnd;
}
@@ -374,16 +399,9 @@ public class MagicDetector implements Detector {
}
if (this.isRegex) {
- int flags = 0;
- if (this.isStringIgnoreCase) {
- flags = Pattern.CASE_INSENSITIVE;
- }
-
- Pattern p = Pattern.compile(new String(this.pattern, UTF_8),
flags);
-
ByteBuffer bb = ByteBuffer.wrap(buffer);
CharBuffer result = ISO_8859_1.decode(bb);
- Matcher m = p.matcher(result);
+ Matcher m = compiledPattern.matcher(result);
boolean match = false;
// Loop until we've covered the entire offset range
diff --git
a/tika-core/src/test/java/org/apache/tika/detect/MagicDetectorTest.java
b/tika-core/src/test/java/org/apache/tika/detect/MagicDetectorTest.java
index 3a86a53b36..6be5fce590 100644
--- a/tika-core/src/test/java/org/apache/tika/detect/MagicDetectorTest.java
+++ b/tika-core/src/test/java/org/apache/tika/detect/MagicDetectorTest.java
@@ -26,6 +26,12 @@ import java.io.BufferedInputStream;
import java.io.ByteArrayInputStream;
import java.io.IOException;
import java.io.InputStream;
+import java.util.ArrayList;
+import java.util.List;
+import java.util.concurrent.ExecutorService;
+import java.util.concurrent.Executors;
+import java.util.concurrent.Future;
+import java.util.concurrent.TimeUnit;
import org.apache.commons.io.IOUtils;
import org.junit.jupiter.api.Test;
@@ -158,6 +164,65 @@ public class MagicDetectorTest {
assertDetect(detector, MediaType.OCTET_STREAM, data2);
}
+ /**
+ * A MagicDetector is built once and reused for the life of the process, so
+ * repeated calls must be independent of each other. Guards the compiled
+ * Pattern against per-call state leaking in.
+ */
+ @Test
+ public void testRegExDetectorRepeatedCallsStable() throws Exception {
+ MediaType html = new MediaType("text", "html");
+ String pattern = "(?s)\\A.{0,1024}\\x3c\\!(?:DOCTYPE|doctype)
(?:HTML|html) ";
+ Detector detector =
+ new MagicDetector(html, pattern.getBytes(US_ASCII), null,
true, 0, 0);
+
+ String match = "<!DOCTYPE HTML PUBLIC \"-//W3C//DTD HTML 4.01//EN\">";
+ String noMatch = "<html><head><title>plain</title></head>";
+
+ for (int i = 0; i < 100; i++) {
+ assertDetect(detector, html, match);
+ assertDetect(detector, MediaType.OCTET_STREAM, noMatch);
+ }
+ }
+
+ /**
+ * MimeTypes shares one MagicDetector instance per magic clause across
every
+ * caller, so the regex path has to be safe to use concurrently.
+ */
+ @Test
+ public void testRegExDetectorConcurrent() throws Exception {
+ MediaType pdf = new MediaType("application", "pdf");
+ Detector detector =
+ new MagicDetector(pdf,
"(?s)\\A.{0,144}%PDF-".getBytes(US_ASCII), null, true, 0, 0);
+
+ byte[] match = "%PDF-1.4\nsome trailing content".getBytes(US_ASCII);
+ byte[] noMatch = "not a pdf at all".getBytes(US_ASCII);
+
+ int threads = 8;
+ int iterations = 200;
+ ExecutorService executor = Executors.newFixedThreadPool(threads);
+ try {
+ List<Future<?>> futures = new ArrayList<>();
+ for (int t = 0; t < threads; t++) {
+ futures.add(executor.submit(() -> {
+ for (int i = 0; i < iterations; i++) {
+ assertEquals(pdf, detector.detect(new
ByteArrayInputStream(match),
+ new Metadata()));
+ assertEquals(MediaType.OCTET_STREAM,
+ detector.detect(new
ByteArrayInputStream(noMatch), new Metadata()));
+ }
+ return null;
+ }));
+ }
+ for (Future<?> future : futures) {
+ // an assertion failure on a worker surfaces here as an
ExecutionException
+ future.get(60, TimeUnit.SECONDS);
+ }
+ } finally {
+ executor.shutdownNow();
+ }
+ }
+
@Test
public void testDetectStreamReadProblems() throws Exception {
byte[] data =
"abcdefghijklmnopqrstuvwxyz0123456789".getBytes(US_ASCII);