[ 
https://issues.apache.org/jira/browse/TIKA-4856?page=com.atlassian.jira.plugin.system.issuetabpanels:comment-tabpanel&focusedCommentId=18109712#comment-18109712
 ] 

ASF GitHub Bot commented on TIKA-4856:
--------------------------------------

dschmidt commented on code in PR #3096:
URL: https://github.com/apache/tika/pull/3096#discussion_r3890247299


##########
tika-server/tika-server-standard/src/test/java/org/apache/tika/server/standard/UnpackerThumbnailTest.java:
##########
@@ -0,0 +1,208 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.server.standard;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+import java.awt.image.BufferedImage;
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.io.InputStream;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.nio.file.Paths;
+import java.util.ArrayList;
+import java.util.Base64;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+import javax.imageio.ImageIO;
+
+import com.fasterxml.jackson.databind.JsonNode;
+import com.fasterxml.jackson.databind.ObjectMapper;
+import jakarta.ws.rs.core.Response;
+import org.apache.cxf.jaxrs.JAXRSServerFactoryBean;
+import org.apache.cxf.jaxrs.client.WebClient;
+import org.apache.cxf.jaxrs.lifecycle.SingletonResourceProvider;
+import org.junit.jupiter.api.Test;
+
+import org.apache.tika.metadata.TikaCoreProperties;
+import org.apache.tika.serialization.config.JsonConfigHelper;
+import org.apache.tika.server.core.CXFTestBase;
+import org.apache.tika.server.core.TikaServerParseExceptionMapper;
+import org.apache.tika.server.core.resource.RecursiveMetadataResource;
+import org.apache.tika.server.core.resource.UnpackerResource;
+import org.apache.tika.server.core.writer.MetadataListMessageBodyWriter;
+
+/**
+ * {@code /unpack/thumbnail} end to end: the document thumbnail comes back as
+ * JSON with its metadata and the image as base64.
+ */
+public class UnpackerThumbnailTest extends CXFTestBase {
+
+    private static final String THUMBNAIL_PATH = "/unpack/thumbnail";
+    private static final String UNPACK_CONFIG_TEMPLATE = 
"/configs/cxf-unpack-test-template.json";
+    private static final ObjectMapper MAPPER = new ObjectMapper();
+
+    private Path unpackTempDir;
+
+    @Override
+    protected void setUpResources(JAXRSServerFactoryBean sf) {
+        sf.setResourceClasses(UnpackerResource.class, 
RecursiveMetadataResource.class);
+        sf.setResourceProvider(UnpackerResource.class,
+                new SingletonResourceProvider(new 
UnpackerResource(tikaResource)));
+        sf.setResourceProvider(RecursiveMetadataResource.class,
+                new SingletonResourceProvider(new 
RecursiveMetadataResource(tikaResource)));
+    }
+
+    @Override
+    protected void setUpProviders(JAXRSServerFactoryBean sf) {
+        List<Object> providers = new ArrayList<>();
+        providers.add(new TikaServerParseExceptionMapper());
+        providers.add(new MetadataListMessageBodyWriter());
+        sf.setProviders(providers);
+    }
+
+    @Override
+    protected InputStream getPipesConfigInputStream() throws IOException {
+        unpackTempDir = 
Files.createTempDirectory("tika-unpack-thumbnail-test-");
+        Path pluginsDir = Paths.get("target/plugins").toAbsolutePath();
+        Map<String, Object> replacements = new HashMap<>();
+        replacements.put("UNPACK_EMITTER_BASE_PATH", 
unpackTempDir.toAbsolutePath().toString());
+        replacements.put("PLUGINS_PATHS", pluginsDir.toString().replace("\\", 
"/"));
+        replacements.put("TIMEOUT_MILLIS", 60000L);
+        JsonNode config = 
JsonConfigHelper.loadFromResource(UNPACK_CONFIG_TEMPLATE,
+                CXFTestBase.class, replacements);
+        return new ByteArrayInputStream(
+                
MAPPER.writeValueAsString(config).getBytes(StandardCharsets.UTF_8));
+    }
+
+    @Override
+    protected Path getUnpackEmitterBasePath() {
+        return unpackTempDir;
+    }
+
+    /**
+     * A stored thumbnail (the docProps thumbnail of a presentation).
+     */
+    @Test
+    public void testStoredThumbnail() throws Exception {
+        JsonNode json = thumbnail("test-documents/testPPTX_Thumbnail.pptx");
+        JsonNode metadata = json.get("metadata");
+        assertEquals("image/jpeg", metadata.get("Content-Type").asText());
+        assertEquals("THUMBNAIL", 
metadata.get("tk:embedded-resource-type").asText());
+        assertEquals("1", metadata.get("tk:embedded-depth").asText());
+        BufferedImage image = decode(json);
+        assertEquals(metadata.get("tiff:ImageWidth").asInt(), 
image.getWidth());
+    }
+
+    /**
+     * A camera raw file: the largest embedded JPEG preview.
+     */
+    @Test
+    public void testRawPreview() throws Exception {
+        JsonNode json = thumbnail("test-documents/testNEF.nef");
+        JsonNode metadata = json.get("metadata");
+        assertEquals("image/jpeg", metadata.get("Content-Type").asText());
+        assertEquals("THUMBNAIL", 
metadata.get("tk:embedded-resource-type").asText());
+        assertEquals(64, decode(json).getWidth());
+    }
+
+    /**
+     * A PDF has no thumbnail; with renderThumbnails the rendering of its first
+     * page stands in, without it there is nothing.
+     */
+    @Test
+    public void testPdfPageRendering() throws Exception {
+        Response plain = WebClient.create(endPoint + THUMBNAIL_PATH)
+                
.put(ClassLoader.getSystemResourceAsStream("test-documents/testPDFTwoTextBoxes.pdf"));
+        assertEquals(204, plain.getStatus());
+
+        JsonNode json = 
thumbnail("test-documents/testPDFTwoTextBoxes.pdf?renderThumbnails=true");
+        JsonNode metadata = json.get("metadata");
+        assertEquals("image/png", metadata.get("Content-Type").asText());
+        assertEquals("RENDERING", 
metadata.get("tk:embedded-resource-type").asText());
+        assertEquals("1", metadata.get("tk:page:number").asText());
+        assertTrue(decode(json).getWidth() > 100);
+    }
+
+    /**
+     * A document without a thumbnail: no content, no error.
+     */
+    @Test
+    public void testNoThumbnail() throws Exception {
+        Response response = WebClient.create(endPoint + THUMBNAIL_PATH)
+                
.put(ClassLoader.getSystemResourceAsStream("test-documents/2pic.docx"));
+        assertEquals(204, response.getStatus());
+    }
+
+    /**
+     * {@code /rmeta?renderThumbnails=true} lays the thumbnail defaults under a
+     * normal metadata request: the first page rendering joins the list, the
+     * text is still extracted. Without the switch nothing is rendered.
+     */
+    @Test
+    public void testRmetaRenderThumbnails() throws Exception {
+        JsonNode plain = rmeta("test-documents/testPDFTwoTextBoxes.pdf", 
false);
+        assertEquals(1, plain.size());
+
+        JsonNode rendered = rmeta("test-documents/testPDFTwoTextBoxes.pdf", 
true);
+        assertEquals(2, rendered.size());
+        
assertTrue(rendered.get(0).get(TikaCoreProperties.TIKA_CONTENT.getName()).asText()
+                .contains("Left column"), rendered.get(0).toString());
+        JsonNode rendering = rendered.get(1);
+        assertEquals("image/png", rendering.get("Content-Type").asText());
+        assertEquals("RENDERING", 
rendering.get("tk:embedded-resource-type").asText());
+        assertEquals("1", rendering.get("tk:page:number").asText());
+        assertTrue(rendering.get("tiff:ImageWidth").asInt() > 100);
+    }
+
+    private JsonNode rmeta(String resource, boolean renderThumbnails) throws 
Exception {
+        Response response = WebClient.create(endPoint + "/rmeta/text"
+                        + (renderThumbnails ? "?renderThumbnails=true" : ""))
+                .put(ClassLoader.getSystemResourceAsStream(resource));
+        assertEquals(200, response.getStatus());
+        return MAPPER.readTree((InputStream) response.getEntity());
+    }
+
+    /**
+     * Sends the file name along, as a client would: raw camera formats are
+     * detected by their extension.
+     */

Review Comment:
   Updated, the comment predates TIKA-4861.



##########
tika-server/tika-server-core/src/main/java/org/apache/tika/server/core/resource/UnpackerResource.java:
##########
@@ -215,15 +273,181 @@ public Response unpackAll(InputStream is, @Context 
HttpHeaders httpHeaders, @Con
     @POST
     @Consumes("multipart/form-data")
     @Produces("application/zip")
-    public Response unpackAllWithConfig(List<Attachment> attachments, @Context 
HttpHeaders httpHeaders, @Context UriInfo info) throws Exception {
+    public Response unpackAllWithConfig(List<Attachment> attachments, @Context 
HttpHeaders httpHeaders, @Context UriInfo info,
+                                     @QueryParam("renderThumbnails") boolean 
renderThumbnails) throws Exception {
         ParseContext pc = tikaResource.createRequestContext();
         Metadata metadata = tikaResource.newRequestMetadata();
         try (TikaInputStream tis = 
tikaResource.setupMultipartConfig(attachments, metadata, pc)) {
             TikaResource.logRequest(LOG, "/unpack/all", metadata);
+            if (renderThumbnails) {
+                //under the request's config, which setupMultipartConfig has 
already merged
+                tikaResource.getThumbnailDefaults().applyTo(pc);
+            }
             return doUnpack(tis, metadata, pc, true);
         }
     }
 
+    /**
+     * Returns the document thumbnail with its metadata (simple PUT).
+     */
+    @jakarta.ws.rs.Path("/thumbnail")
+    @PUT
+    @Produces("application/json")
+    public Response unpackThumbnail(InputStream is, @Context HttpHeaders 
httpHeaders,
+                                    @QueryParam("renderThumbnails") boolean 
renderThumbnails) throws Exception {
+        ParseContext pc = tikaResource.createRequestContext();
+        Metadata metadata = tikaResource.newRequestMetadata();
+        try (TikaInputStream tis = TikaInputStream.get(is)) {
+            fillMetadata(null, metadata, httpHeaders.getRequestHeaders());
+            TikaResource.logRequest(LOG, "/unpack/thumbnail", metadata);
+            return doUnpackThumbnail(tis, metadata, pc, renderThumbnails);
+        }
+    }
+
+    /**
+     * Returns the document thumbnail with its metadata (multipart POST, 
"file" part).
+     */
+    @jakarta.ws.rs.Path("/thumbnail")
+    @POST
+    @Consumes("multipart/form-data")
+    @Produces("application/json")
+    public Response unpackThumbnailMultipart(List<Attachment> attachments, 
@Context HttpHeaders httpHeaders,
+                                             @QueryParam("renderThumbnails") 
boolean renderThumbnails)
+            throws Exception {
+        ParseContext pc = tikaResource.createRequestContext();
+        Metadata metadata = tikaResource.newRequestMetadata();
+        try (TikaInputStream tis = 
tikaResource.setupMultipartConfig(attachments, metadata, pc)) {
+            TikaResource.logRequest(LOG, "/unpack/thumbnail", metadata);
+            return doUnpackThumbnail(tis, metadata, pc, renderThumbnails);
+        }
+    }
+
+    private static final ObjectMapper MAPPER = new ObjectMapper();
+    private static final String METADATA_SUFFIX = ".metadata.json";
+    /**
+     * A thumbnail travels base64-encoded inside a JSON object, so it is
+     * bounded here regardless of the unpack limits; camera previews and
+     * page renderings are a few MB at most.
+     */
+    static final long MAX_THUMBNAIL_BYTES = 32L * 1024 * 1024;
+
+    /**
+     * Parses in unpack mode with the thumbnail configuration, then selects
+     * the thumbnail among the extracted embedded documents.
+     */
+    private Response doUnpackThumbnail(TikaInputStream tis, Metadata metadata, 
ParseContext pc,
+                                       boolean renderThumbnails) throws 
Exception {
+        PipesParsingHelper helper = tikaResource.getPipesParsingHelper();
+        if (helper == null) {
+            throw new WebApplicationException("Pipes-based parsing is not 
enabled", Response.Status.SERVICE_UNAVAILABLE);
+        }
+        configureThumbnailParse(pc, renderThumbnails);
+
+        PipesParsingHelper.UnpackResult result = helper.parseUnpack(tis, 
metadata, pc, false);
+        if (result.zipFile() == null) {
+            throw new WebApplicationException(Response.Status.NO_CONTENT);
+        }
+        try (ZipFile zip = new ZipFile(result.zipFile().toFile())) {
+            Map<String, Metadata> extracted = readExtractedMetadata(zip);
+            Metadata thumbnail = ThumbnailSelector.select(new 
ArrayList<>(extracted.values()));
+            if (thumbnail == null) {
+                throw new WebApplicationException(Response.Status.NO_CONTENT);
+            }
+            String entryName = null;
+            for (Map.Entry<String, Metadata> e : extracted.entrySet()) {
+                if (e.getValue() == thumbnail) {
+                    entryName = e.getKey();
+                }
+            }
+            ZipEntry imageEntry = entryName == null ? null : 
zip.getEntry(entryName);
+            if (imageEntry == null) {
+                throw new WebApplicationException(Response.Status.NO_CONTENT);
+            }
+            if (imageEntry.getSize() > MAX_THUMBNAIL_BYTES) {
+                throw new WebApplicationException("thumbnail larger than " + 
MAX_THUMBNAIL_BYTES + " bytes",
+                        Response.Status.REQUEST_ENTITY_TOO_LARGE);
+            }
+            byte[] image;
+            try (InputStream is = zip.getInputStream(imageEntry)) {
+                //the entry size is a claim; read one byte past the limit to 
know
+                image = is.readNBytes((int) MAX_THUMBNAIL_BYTES + 1);
+            }
+            if (image.length > MAX_THUMBNAIL_BYTES) {
+                throw new WebApplicationException("thumbnail larger than " + 
MAX_THUMBNAIL_BYTES + " bytes",
+                        Response.Status.REQUEST_ENTITY_TOO_LARGE);
+            }
+            StringWriter metadataJson = new StringWriter();
+            JsonMetadata.toJson(thumbnail, metadataJson);
+            ObjectNode root = MAPPER.createObjectNode();
+            root.set("metadata", MAPPER.readTree(metadataJson.toString()));
+            root.put("image", Base64.getEncoder().encodeToString(image));
+            return 
Response.ok(MAPPER.writeValueAsString(root)).type("application/json").build();
+        } finally {
+            result.cleanup();
+        }
+    }
+
+    /**
+     * What only makes sense when the thumbnail is all the caller wants: no
+     * text, no OCR, only THUMBNAIL and RENDERING embedded documents extracted,
+     * together with their metadata, down to the rendering of a thumbnail
+     * (depth 2). With {@code renderThumbnails} the {@link ThumbnailDefaults}
+     * are laid under that, the same switch as on the other endpoints; without
+     * it only stored thumbnails are found. The request's own parser
+     * configuration wins where present.
+     */
+    private void configureThumbnailParse(ParseContext pc, boolean 
renderThumbnails) {
+        //the text is not part of the answer: do not extract it
+        tikaResource.setupContentHandlerFactory(pc, "ignore");
+        ThumbnailDefaults noOcr = ThumbnailDefaults.none()
+                .with("{\"pdf-parser\": {\"ocr\": {\"strategy\": \"NO_OCR\"}}, 
"
+                        + "\"tesseract-ocr-parser\": {\"skipOcr\": true}}");
+        (renderThumbnails ? tikaResource.getThumbnailDefaults().with(noOcr) : 
noOcr).applyTo(pc);

Review Comment:
   Made it a constant.





> /unpack/thumbnail: return the document thumbnail with its metadata
> ------------------------------------------------------------------
>
>                 Key: TIKA-4856
>                 URL: https://issues.apache.org/jira/browse/TIKA-4856
>             Project: Tika
>          Issue Type: New Feature
>            Reporter: Dominik Schmidt
>            Priority: Major
>
> With TIKA-4850 through TIKA-4855 every container format that carries a 
> thumbnail emits it as a THUMBNAIL embedded document, the PDF parser renders 
> pages as RENDERING documents, and the EMF/WMF renderer turns the vector 
> thumbnails of Office documents into raster ones. Getting "the thumbnail of 
> this file" out of that still takes format knowledge on the client: the 
> THUMBNAIL of a Word or Excel file is an EMF/WMF whose usable form is the 
> RENDERING underneath it, a PDF has no THUMBNAIL but a page RENDERING, the 
> THUMBNAIL of a DOCX inside a ZIP is not the ZIP's, and with rendering enabled 
> the picture of an embedded OLE object is a RENDERING too. Plus the request 
> config that switches the renderers on.
> Proposal: POST /unpack/thumbnail next to /unpack and /unpack/all, multipart 
> like them. It runs the usual forked parse in unpack mode with a fixed parse 
> context (PDF page 1 rendered, EMF/WMF rendered) and picks, in this order: the 
> raster THUMBNAIL at depth 1; the rendering of that thumbnail; the depth-1 
> RENDERING of PDF page 1. The endpoint extracts what the document carries; it 
> does not resize, convert or generate previews.
> The response is JSON: the /rmeta metadata object of the selected embedded 
> document, and the image as base64. Thumbnails are small, so the encoding 
> overhead does not matter, and the caller gets type, dimensions, origin 
> (stored thumbnail or rendering, tk:rendering:rendered-by) and path in one 
> round trip without unpacking a zip. 204 when the document has no thumbnail.
> {
>   "metadata": {
>     "Content-Type": "image/png",
>     "Content-Length": "8459",
>     "tiff:ImageWidth": "800",
>     "tiff:ImageLength": "1131",
>     "tk:embedded-resource-type": "RENDERING",
>     "tk:embedded-resource-path": "/thumbnail.emf/thumbnail.png",
>     "tk:embedded-depth": "2",
>     "tk:rendering:rendered-by": "poi-metafile-renderer",
>     "tk:resource-name": "thumbnail.png"
>   },
>   "image": "iVBORw0KGgoAAAANSUhEUgAA..."
> }
> To keep the selection rule short, the metafile renderer could give the 
> rendering of a THUMBNAIL the THUMBNAIL type as well (its 
> tk:rendering:rendered-by tells it apart), so a raster thumbnail is a 
> THUMBNAIL regardless of whether the document stored it as PNG or as EMF.
> What do you think? 



--
This message was sent by Atlassian Jira
(v8.20.10#820010)

Reply via email to