[ 
https://issues.apache.org/jira/browse/TIKA-4936?page=com.atlassian.jira.plugin.system.issuetabpanels:comment-tabpanel&focusedCommentId=18120018#comment-18120018
 ] 

ASF GitHub Bot commented on TIKA-4936:
--------------------------------------

Copilot commented on code in PR #3263:
URL: https://github.com/apache/tika/pull/3263#discussion_r4119860362


##########
tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/executable/PEIconExtractor.java:
##########
@@ -0,0 +1,513 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.parser.executable;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.nio.ByteBuffer;
+import java.nio.ByteOrder;
+import java.nio.channels.Channels;
+import java.nio.channels.FileChannel;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.HashSet;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Set;
+
+import org.apache.commons.io.IOUtils;
+import org.xml.sax.SAXException;
+
+import org.apache.tika.exception.TikaException;
+import org.apache.tika.extractor.EmbeddedDocumentExtractor;
+import org.apache.tika.extractor.EmbeddedDocumentUtil;
+import org.apache.tika.io.EndianUtils;
+import org.apache.tika.io.TikaInputStream;
+import org.apache.tika.metadata.HttpHeaders;
+import org.apache.tika.metadata.Metadata;
+import org.apache.tika.metadata.TikaCoreProperties;
+import org.apache.tika.parser.ParseContext;
+import org.apache.tika.sax.EmbeddedContentHandler;
+import org.apache.tika.sax.XHTMLContentHandler;
+
+/**
+ * Extracts the icons of a PE file (EXE/DLL) from its resource section and
+ * hands each icon group to the {@link EmbeddedDocumentExtractor} as a
+ * standalone <code>.ico</code> file.
+ * <p>
+ * Windows stores an icon as a group resource ({@code RT_GROUP_ICON}) that
+ * lists the individual images, which are stored as {@code RT_ICON} resources.
+ * The {@code .ico} file format is nearly identical to the group resource;
+ * the only difference is that the group refers to its images by resource id
+ * whereas the file refers to them by file offset. This class rebuilds the
+ * file from the two resource types.
+ * <p>
+ * The resource section is read lazily and only as far as the icon data
+ * reaches: a file without icons costs little more than its resource
+ * directory. File-backed input is read through a positioned channel,
+ * anything else by skipping forward, so non-seekable input works too.
+ * Hostile input is contained by bounds checking every offset, capping the
+ * section size, the number of directory entries visited, the icons per group
+ * and the size of a rebuilt icon, and by refusing cycles in the tree.
+ */
+class PEIconExtractor {
+
+    static final String ICON_MIME_TYPE = "image/vnd.microsoft.icon";
+
+    private static final int RT_ICON = 3;
+    private static final int RT_GROUP_ICON = 14;
+
+    private static final int IMAGE_DIRECTORY_ENTRY_RESOURCE = 2;
+    private static final int PE32_MAGIC = 0x10b;
+    private static final int PE32PLUS_MAGIC = 0x20b;
+    private static final int SECTION_HEADER_SIZE = 40;
+    private static final int RESOURCE_DIRECTORY_SIZE = 16;
+    private static final int RESOURCE_DIRECTORY_ENTRY_SIZE = 8;
+    private static final int RESOURCE_DATA_ENTRY_SIZE = 16;
+    private static final int GRP_ICON_DIR_ENTRY_SIZE = 14;
+    private static final int ICON_DIR_ENTRY_SIZE = 16;
+    private static final long HIGH_BIT = 0x80000000L;
+
+    // Sanity limits for hostile input
+    private static final int MAX_SECTIONS = 96; // the PE spec's own limit
+    private static final int MAX_RESOURCE_SECTION_SIZE = 64 * 1024 * 1024;
+    private static final int MAX_RESOURCE_TREE_DEPTH = 3; // type / name / 
language
+    private static final int MAX_DIRECTORY_ENTRIES = 20000; // visited across 
the whole tree
+    private static final int MAX_RESOURCE_NAME_LENGTH = 256;
+    private static final int MAX_ICONS_PER_GROUP = 256;
+    // A genuine icon never outgrows the section it came from
+    private static final long MAX_ICO_SIZE = MAX_RESOURCE_SECTION_SIZE;
+
+    private PEIconExtractor() {
+    }
+
+    /**
+     * Continues reading the PE file directly after the COFF file header and
+     * emits every icon group as an embedded document.
+     *
+     * @param stream      the input positioned right after the 24 byte COFF 
header
+     * @param sizeOptHdrs the SizeOfOptionalHeader field of the COFF header
+     * @param numSections the NumberOfSections field of the COFF header
+     * @param metadata    the PE file's own metadata, receives a warning if the
+     *                    resource tree was too large to walk completely
+     */
+    static void extract(TikaInputStream stream, int sizeOptHdrs, int 
numSections,
+                        XHTMLContentHandler xhtml, Metadata metadata, 
ParseContext context)
+            throws IOException, SAXException, TikaException {
+        if (numSections <= 0 || numSections > MAX_SECTIONS) {
+            return;
+        }
+        // The optional header holds the data directories, of which we
+        // need the one pointing at the resource tree
+        byte[] optHdr = new byte[sizeOptHdrs];
+        IOUtils.readFully(stream, optHdr);
+        int dataDirOffset;
+        switch (sizeOptHdrs >= 2 ? EndianUtils.getUShortLE(optHdr, 0) : 0) {
+            case PE32_MAGIC:
+                dataDirOffset = 96;
+                break;
+            case PE32PLUS_MAGIC:
+                dataDirOffset = 112;
+                break;
+            default:
+                return;
+        }
+        int rsrcEntry = dataDirOffset + IMAGE_DIRECTORY_ENTRY_RESOURCE * 8;
+        if (rsrcEntry + 8 > sizeOptHdrs) {
+            return;
+        }
+        long numDataDirs = EndianUtils.getUIntLE(optHdr, dataDirOffset - 4);
+        if (numDataDirs <= IMAGE_DIRECTORY_ENTRY_RESOURCE) {
+            return;
+        }
+        long rsrcRva = EndianUtils.getUIntLE(optHdr, rsrcEntry);
+        long rsrcSize = EndianUtils.getUIntLE(optHdr, rsrcEntry + 4);
+        if (rsrcRva == 0 || rsrcSize == 0) {
+            return;
+        }
+
+        // The section table tells us where in the file the resource RVA lives
+        byte[] sections = new byte[numSections * SECTION_HEADER_SIZE];
+        IOUtils.readFully(stream, sections);
+        long sectionVa = -1;
+        long sectionRawPtr = -1;
+        long sectionRawSize = -1;
+        for (int i = 0; i < numSections; i++) {
+            int off = i * SECTION_HEADER_SIZE;
+            long va = EndianUtils.getUIntLE(sections, off + 12);
+            long rawSize = EndianUtils.getUIntLE(sections, off + 16);
+            long rawPtr = EndianUtils.getUIntLE(sections, off + 20);
+            if (rsrcRva >= va && rsrcRva < va + rawSize) {
+                sectionVa = va;
+                sectionRawPtr = rawPtr;
+                sectionRawSize = rawSize;
+                break;
+            }
+        }
+        if (sectionVa < 0 || sectionRawSize > MAX_RESOURCE_SECTION_SIZE) {
+            return;
+        }
+
+        InputStream source;
+        if (stream.hasFile()) {
+            FileChannel channel = stream.getFileChannel();
+            if (sectionRawPtr >= channel.size()) {
+                return;
+            }
+            channel.position(sectionRawPtr);
+            source = Channels.newInputStream(channel);
+        } else {
+            // Everything read so far: DOS header up to and including the 
section table
+            long position = stream.getPosition();
+            if (sectionRawPtr < position) {
+                return;
+            }
+            IOUtils.skipFully(stream, sectionRawPtr - position);
+            source = stream;
+        }
+
+        // Offsets inside the resource tree are relative to its root, which
+        // normally but not necessarily sits at the start of the section
+        Resources resources = new Resources(new Section(source, (int) 
sectionRawSize), sectionVa,
+                rsrcRva - sectionVa);
+        readDirectory(resources, resources.rootOffset, 0, new HashSet<>(), 0, 
0, null);
+        if (resources.budget < 0) {
+            EmbeddedDocumentUtil.recordException(new TikaException(
+                    "PE resource directory has more than " + 
MAX_DIRECTORY_ENTRIES +
+                            " entries; icon extraction stopped early"), 
metadata, context);
+        }
+        emitIcons(resources, xhtml, context);
+    }
+
+    /**
+     * Walks the three level resource tree (type / name / language) and
+     * collects every icon and icon group. {@code type}, {@code id} and
+     * {@code name} carry what the levels above have established.
+     */
+    private static void readDirectory(Resources resources, long dirOffset, int 
depth,
+                                      Set<Long> visited, int type, int id, 
String name)
+            throws IOException {
+        if (depth >= MAX_RESOURCE_TREE_DEPTH || !visited.add(dirOffset)) {
+            return;
+        }
+        Section section = resources.section;
+        int dir = section.index(dirOffset, RESOURCE_DIRECTORY_SIZE);
+        if (dir < 0) {
+            return;
+        }
+        int numEntries = EndianUtils.getUShortLE(section.buf, dir + 12) +
+                EndianUtils.getUShortLE(section.buf, dir + 14);
+        for (int i = 0; i < numEntries; i++) {
+            if (--resources.budget < 0) {
+                return;
+            }
+            int entry = section.index(dirOffset + RESOURCE_DIRECTORY_SIZE +
+                    (long) i * RESOURCE_DIRECTORY_ENTRY_SIZE, 
RESOURCE_DIRECTORY_ENTRY_SIZE);
+            if (entry < 0) {
+                return;
+            }
+            long nameField = EndianUtils.getUIntLE(section.buf, entry);
+            long dataField = EndianUtils.getUIntLE(section.buf, entry + 4);
+
+            if (depth == 0) {
+                // Only icons are interesting, and a well-formed tree lists 
each type once
+                if (nameField != RT_ICON && nameField != RT_GROUP_ICON ||
+                        !resources.typesSeen.add((int) nameField)) {
+                    continue;
+                }
+            }
+            boolean named = (nameField & HIGH_BIT) != 0;
+            String entryName = null;
+            if (named) {
+                entryName = readName(section, resources.rootOffset + 
(nameField & ~HIGH_BIT));
+                if (entryName == null) {
+                    continue;
+                }
+            }
+            int entryId = (int) (nameField & 0xffff);
+
+            if ((dataField & HIGH_BIT) != 0) {
+                long subdir = resources.rootOffset + (dataField & ~HIGH_BIT);
+                if (depth == 0) {
+                    readDirectory(resources, subdir, 1, visited, entryId, 0, 
null);
+                } else {
+                    readDirectory(resources, subdir, 2, visited, type, 
entryId, entryName);

Review Comment:
   The depth cap is bypassed for malformed language entries: once `depth > 0`, 
every nested directory is recursed with depth `2`, so a chain below the 
language level can grow until the 20,000-entry budget and potentially overflow 
the Java stack before the advertised depth limit applies. Increment the depth 
on this recursive call so depth-2 directories stop descending.





> Extract icons from PE executables (EXE/DLL) as embedded documents
> -----------------------------------------------------------------
>
>                 Key: TIKA-4936
>                 URL: https://issues.apache.org/jira/browse/TIKA-4936
>             Project: Tika
>          Issue Type: New Feature
>         Environment:  
>  
>  
>  
>            Reporter: Dominik Schmidt
>            Priority: Major
>
> h3. Background
> {{ExecutableParser}} currently only reads the COFF file header of PE files 
> (EXE/DLL) and emits basic metadata (machine type, architecture bits, 
> endianness, created date). The resource section ({{.rsrc}}) is not parsed, so 
> resources such as the application icon are not accessible via Tika.
> h3. Proposal
> Parse the PE resource directory and emit each icon group as an embedded 
> document:
> * Parse optional header, data directories and section table to locate the 
> resource directory (RVA → file offset).
> * Walk the resource tree (type → name/ID → language).
> * For each {{RT_GROUP_ICON}} (type 14), reconstruct a standalone {{.ico}} 
> file from the {{GRPICONDIR}} and the referenced {{RT_ICON}} (type 3) entries: 
> write an {{ICONDIR}} header and replace the 2-byte resource IDs ({{nID}}) 
> with 4-byte image offsets.
> * Pass each reconstructed file to the {{EmbeddedDocumentExtractor}} with:
> ** {{Content-Type}}: {{image/vnd.microsoft.icon}}
> ** {{resourceName}}: e.g. {{icon_<id-or-name>.ico}}
> ** {{embeddedResourceType}}: {{THUMBNAIL}} for the first icon group (the icon 
> shown by Windows Explorer), {{ATTACHMENT}} for all others
> ** the resource language ID
> Single {{RT_ICON}} entries are intentionally not emitted on their own: 
> BMP-based entries are not valid standalone images (no {{BITMAPFILEHEADER}}, 
> double height for the AND mask), and emitting them would duplicate the group 
> data.
> h3. Robustness
> The parser must handle malformed or malicious binaries gracefully:
> * bound the resource tree depth (normally 3 levels) and the number of entries
> * validate all offsets and sizes against the file size
> * guard against cycles in the resource directory
> * failures while extracting resources must not break the existing metadata 
> extraction
> h3. Out of scope (possible follow-ups)
> * {{RT_GROUP_CURSOR}}/{{RT_CURSOR}} → {{.cur}}
> * {{RT_MANIFEST}} as an embedded XML document
> * {{VS_VERSIONINFO}} (product name, file version, company) as metadata
> h3. Acceptance criteria
> * Icons of 32- and 64-bit EXE and DLL test files are extracted as valid 
> {{.ico}} files that are detected as {{image/vnd.microsoft.icon}}.
> * Icon groups containing both PNG- and BMP-encoded entries are supported.
> * Files without a resource section, or without icons, are parsed as before, 
> with no embedded documents.
> * Truncated or corrupted resource sections do not throw and still yield the 
> existing metadata.
> * Unit tests use small, license-compatible test files.



--
This message was sent by Atlassian Jira
(v8.20.10#820010)

Reply via email to