[
https://issues.apache.org/jira/browse/TIKA-4856?page=com.atlassian.jira.plugin.system.issuetabpanels:comment-tabpanel&focusedCommentId=18109710#comment-18109710
]
ASF GitHub Bot commented on TIKA-4856:
--------------------------------------
Copilot commented on code in PR #3096:
URL: https://github.com/apache/tika/pull/3096#discussion_r3890234394
##########
tika-server/tika-server-standard/src/test/java/org/apache/tika/server/standard/UnpackerThumbnailTest.java:
##########
@@ -0,0 +1,208 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.server.standard;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+import java.awt.image.BufferedImage;
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.io.InputStream;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.nio.file.Paths;
+import java.util.ArrayList;
+import java.util.Base64;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+import javax.imageio.ImageIO;
+
+import com.fasterxml.jackson.databind.JsonNode;
+import com.fasterxml.jackson.databind.ObjectMapper;
+import jakarta.ws.rs.core.Response;
+import org.apache.cxf.jaxrs.JAXRSServerFactoryBean;
+import org.apache.cxf.jaxrs.client.WebClient;
+import org.apache.cxf.jaxrs.lifecycle.SingletonResourceProvider;
+import org.junit.jupiter.api.Test;
+
+import org.apache.tika.metadata.TikaCoreProperties;
+import org.apache.tika.serialization.config.JsonConfigHelper;
+import org.apache.tika.server.core.CXFTestBase;
+import org.apache.tika.server.core.TikaServerParseExceptionMapper;
+import org.apache.tika.server.core.resource.RecursiveMetadataResource;
+import org.apache.tika.server.core.resource.UnpackerResource;
+import org.apache.tika.server.core.writer.MetadataListMessageBodyWriter;
+
+/**
+ * {@code /unpack/thumbnail} end to end: the document thumbnail comes back as
+ * JSON with its metadata and the image as base64.
+ */
+public class UnpackerThumbnailTest extends CXFTestBase {
+
+ private static final String THUMBNAIL_PATH = "/unpack/thumbnail";
+ private static final String UNPACK_CONFIG_TEMPLATE =
"/configs/cxf-unpack-test-template.json";
+ private static final ObjectMapper MAPPER = new ObjectMapper();
+
+ private Path unpackTempDir;
+
+ @Override
+ protected void setUpResources(JAXRSServerFactoryBean sf) {
+ sf.setResourceClasses(UnpackerResource.class,
RecursiveMetadataResource.class);
+ sf.setResourceProvider(UnpackerResource.class,
+ new SingletonResourceProvider(new
UnpackerResource(tikaResource)));
+ sf.setResourceProvider(RecursiveMetadataResource.class,
+ new SingletonResourceProvider(new
RecursiveMetadataResource(tikaResource)));
+ }
+
+ @Override
+ protected void setUpProviders(JAXRSServerFactoryBean sf) {
+ List<Object> providers = new ArrayList<>();
+ providers.add(new TikaServerParseExceptionMapper());
+ providers.add(new MetadataListMessageBodyWriter());
+ sf.setProviders(providers);
+ }
+
+ @Override
+ protected InputStream getPipesConfigInputStream() throws IOException {
+ unpackTempDir =
Files.createTempDirectory("tika-unpack-thumbnail-test-");
+ Path pluginsDir = Paths.get("target/plugins").toAbsolutePath();
+ Map<String, Object> replacements = new HashMap<>();
+ replacements.put("UNPACK_EMITTER_BASE_PATH",
unpackTempDir.toAbsolutePath().toString());
+ replacements.put("PLUGINS_PATHS", pluginsDir.toString().replace("\\",
"/"));
+ replacements.put("TIMEOUT_MILLIS", 60000L);
+ JsonNode config =
JsonConfigHelper.loadFromResource(UNPACK_CONFIG_TEMPLATE,
+ CXFTestBase.class, replacements);
+ return new ByteArrayInputStream(
+
MAPPER.writeValueAsString(config).getBytes(StandardCharsets.UTF_8));
+ }
+
+ @Override
+ protected Path getUnpackEmitterBasePath() {
+ return unpackTempDir;
+ }
+
+ /**
+ * A stored thumbnail (the docProps thumbnail of a presentation).
+ */
+ @Test
+ public void testStoredThumbnail() throws Exception {
+ JsonNode json = thumbnail("test-documents/testPPTX_Thumbnail.pptx");
+ JsonNode metadata = json.get("metadata");
+ assertEquals("image/jpeg", metadata.get("Content-Type").asText());
+ assertEquals("THUMBNAIL",
metadata.get("tk:embedded-resource-type").asText());
+ assertEquals("1", metadata.get("tk:embedded-depth").asText());
+ BufferedImage image = decode(json);
+ assertEquals(metadata.get("tiff:ImageWidth").asInt(),
image.getWidth());
+ }
+
+ /**
+ * A camera raw file: the largest embedded JPEG preview.
+ */
+ @Test
+ public void testRawPreview() throws Exception {
+ JsonNode json = thumbnail("test-documents/testNEF.nef");
+ JsonNode metadata = json.get("metadata");
+ assertEquals("image/jpeg", metadata.get("Content-Type").asText());
+ assertEquals("THUMBNAIL",
metadata.get("tk:embedded-resource-type").asText());
+ assertEquals(64, decode(json).getWidth());
+ }
+
+ /**
+ * A PDF has no thumbnail; with renderThumbnails the rendering of its first
+ * page stands in, without it there is nothing.
+ */
+ @Test
+ public void testPdfPageRendering() throws Exception {
+ Response plain = WebClient.create(endPoint + THUMBNAIL_PATH)
+
.put(ClassLoader.getSystemResourceAsStream("test-documents/testPDFTwoTextBoxes.pdf"));
+ assertEquals(204, plain.getStatus());
+
+ JsonNode json =
thumbnail("test-documents/testPDFTwoTextBoxes.pdf?renderThumbnails=true");
+ JsonNode metadata = json.get("metadata");
+ assertEquals("image/png", metadata.get("Content-Type").asText());
+ assertEquals("RENDERING",
metadata.get("tk:embedded-resource-type").asText());
+ assertEquals("1", metadata.get("tk:page:number").asText());
+ assertTrue(decode(json).getWidth() > 100);
+ }
+
+ /**
+ * A document without a thumbnail: no content, no error.
+ */
+ @Test
+ public void testNoThumbnail() throws Exception {
+ Response response = WebClient.create(endPoint + THUMBNAIL_PATH)
+
.put(ClassLoader.getSystemResourceAsStream("test-documents/2pic.docx"));
+ assertEquals(204, response.getStatus());
+ }
+
+ /**
+ * {@code /rmeta?renderThumbnails=true} lays the thumbnail defaults under a
+ * normal metadata request: the first page rendering joins the list, the
+ * text is still extracted. Without the switch nothing is rendered.
+ */
+ @Test
+ public void testRmetaRenderThumbnails() throws Exception {
+ JsonNode plain = rmeta("test-documents/testPDFTwoTextBoxes.pdf",
false);
+ assertEquals(1, plain.size());
+
+ JsonNode rendered = rmeta("test-documents/testPDFTwoTextBoxes.pdf",
true);
+ assertEquals(2, rendered.size());
+
assertTrue(rendered.get(0).get(TikaCoreProperties.TIKA_CONTENT.getName()).asText()
+ .contains("Left column"), rendered.get(0).toString());
+ JsonNode rendering = rendered.get(1);
+ assertEquals("image/png", rendering.get("Content-Type").asText());
+ assertEquals("RENDERING",
rendering.get("tk:embedded-resource-type").asText());
+ assertEquals("1", rendering.get("tk:page:number").asText());
+ assertTrue(rendering.get("tiff:ImageWidth").asInt() > 100);
+ }
+
+ private JsonNode rmeta(String resource, boolean renderThumbnails) throws
Exception {
+ Response response = WebClient.create(endPoint + "/rmeta/text"
+ + (renderThumbnails ? "?renderThumbnails=true" : ""))
+ .put(ClassLoader.getSystemResourceAsStream(resource));
+ assertEquals(200, response.getStatus());
+ return MAPPER.readTree((InputStream) response.getEntity());
+ }
+
+ /**
+ * Sends the file name along, as a client would: raw camera formats are
+ * detected by their extension.
+ */
Review Comment:
This Javadoc is now inaccurate: TIFF-based raw camera formats (e.g., NEF)
can be detected by content (see RawTiffDetector), so they are not necessarily
detected by extension. Updating the comment will prevent confusion about why
the test sends a filename.
##########
tika-server/tika-server-core/src/main/java/org/apache/tika/server/core/resource/UnpackerResource.java:
##########
@@ -215,15 +273,181 @@ public Response unpackAll(InputStream is, @Context
HttpHeaders httpHeaders, @Con
@POST
@Consumes("multipart/form-data")
@Produces("application/zip")
- public Response unpackAllWithConfig(List<Attachment> attachments, @Context
HttpHeaders httpHeaders, @Context UriInfo info) throws Exception {
+ public Response unpackAllWithConfig(List<Attachment> attachments, @Context
HttpHeaders httpHeaders, @Context UriInfo info,
+ @QueryParam("renderThumbnails") boolean
renderThumbnails) throws Exception {
ParseContext pc = tikaResource.createRequestContext();
Metadata metadata = tikaResource.newRequestMetadata();
try (TikaInputStream tis =
tikaResource.setupMultipartConfig(attachments, metadata, pc)) {
TikaResource.logRequest(LOG, "/unpack/all", metadata);
+ if (renderThumbnails) {
+ //under the request's config, which setupMultipartConfig has
already merged
+ tikaResource.getThumbnailDefaults().applyTo(pc);
+ }
return doUnpack(tis, metadata, pc, true);
}
}
+ /**
+ * Returns the document thumbnail with its metadata (simple PUT).
+ */
+ @jakarta.ws.rs.Path("/thumbnail")
+ @PUT
+ @Produces("application/json")
+ public Response unpackThumbnail(InputStream is, @Context HttpHeaders
httpHeaders,
+ @QueryParam("renderThumbnails") boolean
renderThumbnails) throws Exception {
+ ParseContext pc = tikaResource.createRequestContext();
+ Metadata metadata = tikaResource.newRequestMetadata();
+ try (TikaInputStream tis = TikaInputStream.get(is)) {
+ fillMetadata(null, metadata, httpHeaders.getRequestHeaders());
+ TikaResource.logRequest(LOG, "/unpack/thumbnail", metadata);
+ return doUnpackThumbnail(tis, metadata, pc, renderThumbnails);
+ }
+ }
+
+ /**
+ * Returns the document thumbnail with its metadata (multipart POST,
"file" part).
+ */
+ @jakarta.ws.rs.Path("/thumbnail")
+ @POST
+ @Consumes("multipart/form-data")
+ @Produces("application/json")
+ public Response unpackThumbnailMultipart(List<Attachment> attachments,
@Context HttpHeaders httpHeaders,
+ @QueryParam("renderThumbnails")
boolean renderThumbnails)
+ throws Exception {
+ ParseContext pc = tikaResource.createRequestContext();
+ Metadata metadata = tikaResource.newRequestMetadata();
+ try (TikaInputStream tis =
tikaResource.setupMultipartConfig(attachments, metadata, pc)) {
+ TikaResource.logRequest(LOG, "/unpack/thumbnail", metadata);
+ return doUnpackThumbnail(tis, metadata, pc, renderThumbnails);
+ }
+ }
+
+ private static final ObjectMapper MAPPER = new ObjectMapper();
+ private static final String METADATA_SUFFIX = ".metadata.json";
+ /**
+ * A thumbnail travels base64-encoded inside a JSON object, so it is
+ * bounded here regardless of the unpack limits; camera previews and
+ * page renderings are a few MB at most.
+ */
+ static final long MAX_THUMBNAIL_BYTES = 32L * 1024 * 1024;
+
+ /**
+ * Parses in unpack mode with the thumbnail configuration, then selects
+ * the thumbnail among the extracted embedded documents.
+ */
+ private Response doUnpackThumbnail(TikaInputStream tis, Metadata metadata,
ParseContext pc,
+ boolean renderThumbnails) throws
Exception {
+ PipesParsingHelper helper = tikaResource.getPipesParsingHelper();
+ if (helper == null) {
+ throw new WebApplicationException("Pipes-based parsing is not
enabled", Response.Status.SERVICE_UNAVAILABLE);
+ }
+ configureThumbnailParse(pc, renderThumbnails);
+
+ PipesParsingHelper.UnpackResult result = helper.parseUnpack(tis,
metadata, pc, false);
+ if (result.zipFile() == null) {
+ throw new WebApplicationException(Response.Status.NO_CONTENT);
+ }
+ try (ZipFile zip = new ZipFile(result.zipFile().toFile())) {
+ Map<String, Metadata> extracted = readExtractedMetadata(zip);
+ Metadata thumbnail = ThumbnailSelector.select(new
ArrayList<>(extracted.values()));
+ if (thumbnail == null) {
+ throw new WebApplicationException(Response.Status.NO_CONTENT);
+ }
+ String entryName = null;
+ for (Map.Entry<String, Metadata> e : extracted.entrySet()) {
+ if (e.getValue() == thumbnail) {
+ entryName = e.getKey();
+ }
+ }
+ ZipEntry imageEntry = entryName == null ? null :
zip.getEntry(entryName);
+ if (imageEntry == null) {
+ throw new WebApplicationException(Response.Status.NO_CONTENT);
+ }
+ if (imageEntry.getSize() > MAX_THUMBNAIL_BYTES) {
+ throw new WebApplicationException("thumbnail larger than " +
MAX_THUMBNAIL_BYTES + " bytes",
+ Response.Status.REQUEST_ENTITY_TOO_LARGE);
+ }
+ byte[] image;
+ try (InputStream is = zip.getInputStream(imageEntry)) {
+ //the entry size is a claim; read one byte past the limit to
know
+ image = is.readNBytes((int) MAX_THUMBNAIL_BYTES + 1);
+ }
+ if (image.length > MAX_THUMBNAIL_BYTES) {
+ throw new WebApplicationException("thumbnail larger than " +
MAX_THUMBNAIL_BYTES + " bytes",
+ Response.Status.REQUEST_ENTITY_TOO_LARGE);
+ }
+ StringWriter metadataJson = new StringWriter();
+ JsonMetadata.toJson(thumbnail, metadataJson);
+ ObjectNode root = MAPPER.createObjectNode();
+ root.set("metadata", MAPPER.readTree(metadataJson.toString()));
+ root.put("image", Base64.getEncoder().encodeToString(image));
+ return
Response.ok(MAPPER.writeValueAsString(root)).type("application/json").build();
+ } finally {
+ result.cleanup();
+ }
+ }
+
+ /**
+ * What only makes sense when the thumbnail is all the caller wants: no
+ * text, no OCR, only THUMBNAIL and RENDERING embedded documents extracted,
+ * together with their metadata, down to the rendering of a thumbnail
+ * (depth 2). With {@code renderThumbnails} the {@link ThumbnailDefaults}
+ * are laid under that, the same switch as on the other endpoints; without
+ * it only stored thumbnails are found. The request's own parser
+ * configuration wins where present.
+ */
+ private void configureThumbnailParse(ParseContext pc, boolean
renderThumbnails) {
+ //the text is not part of the answer: do not extract it
+ tikaResource.setupContentHandlerFactory(pc, "ignore");
+ ThumbnailDefaults noOcr = ThumbnailDefaults.none()
+ .with("{\"pdf-parser\": {\"ocr\": {\"strategy\": \"NO_OCR\"}},
"
+ + "\"tesseract-ocr-parser\": {\"skipOcr\": true}}");
+ (renderThumbnails ? tikaResource.getThumbnailDefaults().with(noOcr) :
noOcr).applyTo(pc);
Review Comment:
`configureThumbnailParse` rebuilds `noOcr` by parsing JSON on every request.
This adds avoidable Jackson parsing/allocation to a hot path;
`ThumbnailDefaults` is immutable and can be cached as a static constant.
> /unpack/thumbnail: return the document thumbnail with its metadata
> ------------------------------------------------------------------
>
> Key: TIKA-4856
> URL: https://issues.apache.org/jira/browse/TIKA-4856
> Project: Tika
> Issue Type: New Feature
> Reporter: Dominik Schmidt
> Priority: Major
>
> With TIKA-4850 through TIKA-4855 every container format that carries a
> thumbnail emits it as a THUMBNAIL embedded document, the PDF parser renders
> pages as RENDERING documents, and the EMF/WMF renderer turns the vector
> thumbnails of Office documents into raster ones. Getting "the thumbnail of
> this file" out of that still takes format knowledge on the client: the
> THUMBNAIL of a Word or Excel file is an EMF/WMF whose usable form is the
> RENDERING underneath it, a PDF has no THUMBNAIL but a page RENDERING, the
> THUMBNAIL of a DOCX inside a ZIP is not the ZIP's, and with rendering enabled
> the picture of an embedded OLE object is a RENDERING too. Plus the request
> config that switches the renderers on.
> Proposal: POST /unpack/thumbnail next to /unpack and /unpack/all, multipart
> like them. It runs the usual forked parse in unpack mode with a fixed parse
> context (PDF page 1 rendered, EMF/WMF rendered) and picks, in this order: the
> raster THUMBNAIL at depth 1; the rendering of that thumbnail; the depth-1
> RENDERING of PDF page 1. The endpoint extracts what the document carries; it
> does not resize, convert or generate previews.
> The response is JSON: the /rmeta metadata object of the selected embedded
> document, and the image as base64. Thumbnails are small, so the encoding
> overhead does not matter, and the caller gets type, dimensions, origin
> (stored thumbnail or rendering, tk:rendering:rendered-by) and path in one
> round trip without unpacking a zip. 204 when the document has no thumbnail.
> {
> "metadata": {
> "Content-Type": "image/png",
> "Content-Length": "8459",
> "tiff:ImageWidth": "800",
> "tiff:ImageLength": "1131",
> "tk:embedded-resource-type": "RENDERING",
> "tk:embedded-resource-path": "/thumbnail.emf/thumbnail.png",
> "tk:embedded-depth": "2",
> "tk:rendering:rendered-by": "poi-metafile-renderer",
> "tk:resource-name": "thumbnail.png"
> },
> "image": "iVBORw0KGgoAAAANSUhEUgAA..."
> }
> To keep the selection rule short, the metafile renderer could give the
> rendering of a THUMBNAIL the THUMBNAIL type as well (its
> tk:rendering:rendered-by tells it apart), so a raster thumbnail is a
> THUMBNAIL regardless of whether the document stored it as PNG or as EMF.
> What do you think?
--
This message was sent by Atlassian Jira
(v8.20.10#820010)