This is an automated email from the ASF dual-hosted git repository.
tballison pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/tika.git
The following commit(s) were added to refs/heads/main by this push:
new 25869dec99 TIKA-4956: GeoGebraParser reads its JSON with jackson-core
only (#3302)
25869dec99 is described below
commit 25869dec99f64804315a36d54b64c87ad1f85495
Author: Tim Allison <[email protected]>
AuthorDate: Tue Oct 6 11:47:28 2026 -0400
TIKA-4956: GeoGebraParser reads its JSON with jackson-core only (#3302)
Co-authored-by: Tilman Hausherr <[email protected]>
---
CHANGES.txt | 5 +
docs/modules/ROOT/pages/using-tika/osgi.adoc | 2 +-
tika-bundles/tika-bundle-standard/pom.xml | 1 -
tika-bundles/tika-bundle-standard/test-bundles.xml | 2 -
.../tika-parser-cad-module/pom.xml | 4 -
.../tika-parser-miscoffice-module/pom.xml | 2 +-
.../apache/tika/parser/geogebra/GeoGebraJson.java | 124 +++++++++++++++++++++
.../tika/parser/geogebra/GeoGebraParser.java | 16 +--
.../tika/parser/geogebra/GeoGebraXMLHandler.java | 13 +--
.../tika/parser/geogebra/GeoGebraJsonTest.java | 84 ++++++++++++++
10 files changed, 223 insertions(+), 30 deletions(-)
diff --git a/CHANGES.txt b/CHANGES.txt
index 8864254b45..13cf6d8b60 100644
--- a/CHANGES.txt
+++ b/CHANGES.txt
@@ -1,5 +1,10 @@
Release 4.2.0 - unreleased
+ * GeoGebraParser reads structure.json and inline text content with
jackson-core's
+ streaming parser; tika-parser-miscoffice-module and
tika-parser-cad-module no
+ longer depend on jackson-databind, and OSGi deployments of
tika-bundle-standard
+ need only jackson-core (TIKA-4956).
+
* tika-bundle-standard no longer embeds dependencies that are OSGi bundles
themselves (commons-*, pdfbox, fontbox, bouncycastle, jsoup, asm, xz,
xmpcore,
dd-plist); deploy them alongside it (TIKA-4934). Its imports of those
packages,
diff --git a/docs/modules/ROOT/pages/using-tika/osgi.adoc
b/docs/modules/ROOT/pages/using-tika/osgi.adoc
index bf8d35be28..5aefe5d909 100644
--- a/docs/modules/ROOT/pages/using-tika/osgi.adoc
+++ b/docs/modules/ROOT/pages/using-tika/osgi.adoc
@@ -48,7 +48,7 @@ Tika bundles so they can be shared with the rest of your
container and updated o
|`org.bouncycastle:bcprov-jdk18on`, `bcpkix-jdk18on`, `bcutil-jdk18on`,
`bcjmail-jdk18on`
|Jackson
-|`com.fasterxml.jackson.core:jackson-databind`, `jackson-core`,
`jackson-annotations`
+|`com.fasterxml.jackson.core:jackson-core`
|CommonMark
|`org.commonmark:commonmark`, `commonmark-ext-gfm-tables`,
`commonmark-ext-gfm-strikethrough`
diff --git a/tika-bundles/tika-bundle-standard/pom.xml
b/tika-bundles/tika-bundle-standard/pom.xml
index 4af63ff2b7..d7d0dda0aa 100644
--- a/tika-bundles/tika-bundle-standard/pom.xml
+++ b/tika-bundles/tika-bundle-standard/pom.xml
@@ -172,7 +172,6 @@
com.adobe.internal.xmp;com.adobe.internal.xmp.*,
com.dd.plist,
com.fasterxml.jackson.core;com.fasterxml.jackson.core.*,
- com.fasterxml.jackson.databind,
org.apache.commons.codec;org.apache.commons.codec.*,
org.apache.commons.collections4;org.apache.commons.collections4.*,
org.apache.commons.compress;org.apache.commons.compress.*,
diff --git a/tika-bundles/tika-bundle-standard/test-bundles.xml
b/tika-bundles/tika-bundle-standard/test-bundles.xml
index 4ddc350534..e838a3038d 100644
--- a/tika-bundles/tika-bundle-standard/test-bundles.xml
+++ b/tika-bundles/tika-bundle-standard/test-bundles.xml
@@ -31,9 +31,7 @@
<include>org.apache.tika:tika-bundle-standard</include>
<!-- Dependencies that are OSGi bundles themselves (not embedded) -->
<include>com.adobe.xmp:xmpcore</include>
- <include>com.fasterxml.jackson.core:jackson-annotations</include>
<include>com.fasterxml.jackson.core:jackson-core</include>
- <include>com.fasterxml.jackson.core:jackson-databind</include>
<include>com.googlecode.plist:dd-plist</include>
<include>commons-codec:commons-codec</include>
<include>commons-io:commons-io</include>
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-cad-module/pom.xml
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-cad-module/pom.xml
index 5de4666a15..e7178ab794 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-cad-module/pom.xml
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-cad-module/pom.xml
@@ -40,10 +40,6 @@
<groupId>com.fasterxml.jackson.core</groupId>
<artifactId>jackson-core</artifactId>
</dependency>
- <dependency>
- <groupId>com.fasterxml.jackson.core</groupId>
- <artifactId>jackson-databind</artifactId>
- </dependency>
</dependencies>
<properties>
<!-- TIKA-4816 metadata string-key ban, main sources; see tika-parent -->
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/pom.xml
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/pom.xml
index 4841143b88..0c3cdea951 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/pom.xml
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/pom.xml
@@ -88,7 +88,7 @@
<!-- for the GeoGebra structure.json and inline text content -->
<dependency>
<groupId>com.fasterxml.jackson.core</groupId>
- <artifactId>jackson-databind</artifactId>
+ <artifactId>jackson-core</artifactId>
</dependency>
</dependencies>
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraJson.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraJson.java
new file mode 100644
index 0000000000..32f1eb1029
--- /dev/null
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraJson.java
@@ -0,0 +1,124 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.parser.geogebra;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.util.ArrayDeque;
+import java.util.ArrayList;
+import java.util.Arrays;
+import java.util.Deque;
+import java.util.List;
+
+import com.fasterxml.jackson.core.JsonFactory;
+import com.fasterxml.jackson.core.JsonParser;
+import com.fasterxml.jackson.core.JsonToken;
+
+/**
+ * Streaming readers for the two bits of JSON in a GeoGebra file. Both read
+ * with jackson-core only; the parsers do not need a tree.
+ */
+final class GeoGebraJson {
+
+ private static final JsonFactory JSON_FACTORY = new JsonFactory();
+
+ /**
+ * The containers an element id sits in: {@code {"chapters":[{"pages":
+ * [{"elements":[{"id":...}]}]}]}}. A container opened as an array element
+ * or at the root has no field name.
+ */
+ private static final List<String> ELEMENT_PATH =
+ Arrays.asList(null, "chapters", null, "pages", null, "elements",
null);
+
+ private GeoGebraJson() {
+ }
+
+ /**
+ * Returns the {@code id}s of the elements of a {@code structure.json},
+ * in document order.
+ *
+ * @throws IOException if the stream is not well-formed JSON
+ */
+ static List<String> elementIds(InputStream is) throws IOException {
+ List<String> ids = new ArrayList<>();
+ try (JsonParser parser = JSON_FACTORY.createParser(is)) {
+ Deque<String> path = new ArrayDeque<>();
+ String fieldName = null;
+ for (JsonToken t = parser.nextToken(); t != null; t =
parser.nextToken()) {
+ if (t == JsonToken.FIELD_NAME) {
+ fieldName = parser.currentName();
+ } else if (t.isStructStart()) {
+ path.addLast(fieldName == null ? "" : fieldName);
+ fieldName = null;
+ } else if (t.isStructEnd()) {
+ path.removeLast();
+ } else {
+ if ("id".equals(fieldName) && t.isScalarValue() &&
atElement(path)) {
+ ids.add(parser.getText());
+ }
+ fieldName = null;
+ }
+ }
+ }
+ return ids;
+ }
+
+ private static boolean atElement(Deque<String> path) {
+ if (path.size() != ELEMENT_PATH.size()) {
+ return false;
+ }
+ int i = 0;
+ for (String name : path) {
+ String expected = ELEMENT_PATH.get(i++);
+ if (!(expected == null ? "" : expected).equals(name)) {
+ return false;
+ }
+ }
+ return true;
+ }
+
+ /**
+ * Returns the values of every {@code text} field, at any depth, in
+ * document order: the text runs of a {@code content} value such as
+ * {@code [{"text":"Hello\n"}]}. A {@code text} field whose value is null,
+ * an object or an array contributes nothing and is not descended into.
+ *
+ * @throws IOException if the string is not well-formed JSON
+ */
+ static List<String> textValues(String json) throws IOException {
+ List<String> texts = new ArrayList<>();
+ try (JsonParser parser = JSON_FACTORY.createParser(json)) {
+ for (JsonToken t = parser.nextToken(); t != null; t =
parser.nextToken()) {
+ if (t == JsonToken.FIELD_NAME &&
"text".equals(parser.currentName())) {
+ t = parser.nextToken();
+ if (t == null) {
+ break;
+ }
+ if (t == JsonToken.VALUE_NULL) {
+ continue;
+ }
+ if (t.isScalarValue()) {
+ texts.add(parser.getText());
+ } else {
+ parser.skipChildren();
+ }
+ }
+ }
+ }
+ return texts;
+ }
+}
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraParser.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraParser.java
index e8033ec742..2aea65f186 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraParser.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraParser.java
@@ -31,8 +31,6 @@ import java.util.Set;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
-import com.fasterxml.jackson.databind.JsonNode;
-import com.fasterxml.jackson.databind.ObjectMapper;
import org.apache.commons.compress.archivers.zip.ZipArchiveEntry;
import org.apache.commons.compress.archivers.zip.ZipFile;
import org.xml.sax.ContentHandler;
@@ -166,8 +164,6 @@ public class GeoGebraParser implements Parser {
*/
private static final long MAX_STRUCTURE_JSON_LENGTH = 1024 * 1024;
- static final ObjectMapper OBJECT_MAPPER = new ObjectMapper();
-
@Override
public Set<MediaType> getSupportedTypes(ParseContext context) {
return SUPPORTED_TYPES;
@@ -259,15 +255,9 @@ public class GeoGebraParser implements Parser {
Set<String> knownSlideIds = new HashSet<>(numericallySorted);
try (InputStream is = new
BoundedInputStream(MAX_STRUCTURE_JSON_LENGTH,
zipFile.getInputStream(structure))) {
- JsonNode root = OBJECT_MAPPER.readTree(is);
- for (JsonNode chapter : root.path("chapters")) {
- for (JsonNode page : chapter.path("pages")) {
- for (JsonNode element : page.path("elements")) {
- String id = element.path("id").asText("");
- if (knownSlideIds.contains(id)) {
- ordered.add(id);
- }
- }
+ for (String id : GeoGebraJson.elementIds(is)) {
+ if (knownSlideIds.contains(id)) {
+ ordered.add(id);
}
}
} catch (IOException e) {
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraXMLHandler.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraXMLHandler.java
index e5d65a7bd9..2db0fa5498 100644
---
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraXMLHandler.java
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/main/java/org/apache/tika/parser/geogebra/GeoGebraXMLHandler.java
@@ -22,7 +22,6 @@ import java.util.List;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
-import com.fasterxml.jackson.databind.JsonNode;
import org.xml.sax.Attributes;
import org.xml.sax.SAXException;
import org.xml.sax.helpers.DefaultHandler;
@@ -53,7 +52,7 @@ class GeoGebraXMLHandler extends DefaultHandler {
/**
* Longest content JSON that is parsed; a real inline text, table or mind
- * map is a few kilobytes, anything far beyond that is not worth a tree.
+ * map is a few kilobytes.
*/
private static final int MAX_CONTENT_LENGTH = 1024 * 1024;
@@ -164,16 +163,14 @@ class GeoGebraXMLHandler extends DefaultHandler {
//not a JSON document; a plain string carries no text runs
return;
}
- JsonNode root;
+ StringBuilder sb = new StringBuilder();
try {
- root = GeoGebraParser.OBJECT_MAPPER.readTree(trimmed);
+ for (String text : GeoGebraJson.textValues(trimmed)) {
+ sb.append(text);
+ }
} catch (IOException e) {
return;
}
- StringBuilder sb = new StringBuilder();
- for (String text : root.findValuesAsText("text")) {
- sb.append(text);
- }
for (String line : sb.toString().split("\r\n|[\r\n]")) {
paragraph(line);
}
diff --git
a/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/test/java/org/apache/tika/parser/geogebra/GeoGebraJsonTest.java
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/test/java/org/apache/tika/parser/geogebra/GeoGebraJsonTest.java
new file mode 100644
index 0000000000..85eecbde6b
--- /dev/null
+++
b/tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-miscoffice-module/src/test/java/org/apache/tika/parser/geogebra/GeoGebraJsonTest.java
@@ -0,0 +1,84 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.parser.geogebra;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.nio.charset.StandardCharsets;
+import java.util.Arrays;
+import java.util.Collections;
+import java.util.List;
+
+import org.junit.jupiter.api.Test;
+
+public class GeoGebraJsonTest {
+
+ private static List<String> ids(String json) throws IOException {
+ return GeoGebraJson.elementIds(
+ new
ByteArrayInputStream(json.getBytes(StandardCharsets.UTF_8)));
+ }
+
+ @Test
+ public void testElementIdsInDocumentOrder() throws Exception {
+ assertEquals(Arrays.asList("_slide1", "_slide0", "_slide2"), ids(
+ "{\"id\":\"doc\",\"chapters\":[{\"id\":\"c1\",\"pages\":["
+ +
"{\"id\":\"p1\",\"elements\":[{\"id\":\"_slide1\",\"type\":\"x\"},"
+ + "{\"id\":\"_slide0\"}]}]},"
+ +
"{\"pages\":[{\"elements\":[{\"id\":\"_slide2\"}]}]}]}"));
+ }
+
+ @Test
+ public void testIdsOutsideElementsIgnored() throws Exception {
+ //ids on the document, chapter and page, nested deeper, and in other
arrays
+ assertEquals(Collections.emptyList(), ids(
+
"{\"id\":\"doc\",\"chapters\":[{\"id\":\"c\",\"pages\":[{\"id\":\"p\","
+ + "\"elements\":[{\"meta\":{\"id\":\"deep\"}}],"
+ + "\"other\":[{\"id\":\"o\"}]}]}]}"));
+ assertEquals(Collections.emptyList(), ids("{}"));
+ assertEquals(Collections.emptyList(), ids("[]"));
+ }
+
+ @Test
+ public void testMalformedStructureThrows() {
+ assertThrows(IOException.class, () -> ids("not json"));
+ assertThrows(IOException.class, () ->
ids("{\"chapters\":[{\"pages\":["));
+ }
+
+ @Test
+ public void testTextValuesAtAnyDepth() throws Exception {
+ assertEquals(Arrays.asList("Hello\n", "World", "nested", "3"),
+
GeoGebraJson.textValues("[{\"text\":\"Hello\\n\",\"bold\":true},"
+ +
"{\"text\":\"World\",\"children\":[{\"text\":\"nested\"}]},"
+ + "{\"rows\":[[{\"text\":3}]]}]"));
+ }
+
+ @Test
+ public void testContainerAndNullTextSkipped() throws Exception {
+ //a text field holding an object or array is skipped whole, nulls add
nothing
+ assertEquals(Collections.singletonList("after"),
GeoGebraJson.textValues(
+
"[{\"text\":{\"text\":\"inner\"}},{\"text\":[{\"text\":\"inner\"}]},"
+ + "{\"text\":null},{\"text\":\"after\"}]"));
+ }
+
+ @Test
+ public void testMalformedContentThrows() {
+ assertThrows(IOException.class, () ->
GeoGebraJson.textValues("[{\"text\":\"x\"}"));
+ }
+}