[
https://issues.apache.org/jira/browse/TIKA-4936?page=com.atlassian.jira.plugin.system.issuetabpanels:comment-tabpanel&focusedCommentId=18120018#comment-18120018
]
ASF GitHub Bot commented on TIKA-4936:
--------------------------------------
Copilot commented on code in PR #3263:
URL: https://github.com/apache/tika/pull/3263#discussion_r4119860362
##########
tika-parsers/tika-parsers-standard/tika-parsers-standard-modules/tika-parser-code-module/src/main/java/org/apache/tika/parser/executable/PEIconExtractor.java:
##########
@@ -0,0 +1,513 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.tika.parser.executable;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.nio.ByteBuffer;
+import java.nio.ByteOrder;
+import java.nio.channels.Channels;
+import java.nio.channels.FileChannel;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.HashSet;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Set;
+
+import org.apache.commons.io.IOUtils;
+import org.xml.sax.SAXException;
+
+import org.apache.tika.exception.TikaException;
+import org.apache.tika.extractor.EmbeddedDocumentExtractor;
+import org.apache.tika.extractor.EmbeddedDocumentUtil;
+import org.apache.tika.io.EndianUtils;
+import org.apache.tika.io.TikaInputStream;
+import org.apache.tika.metadata.HttpHeaders;
+import org.apache.tika.metadata.Metadata;
+import org.apache.tika.metadata.TikaCoreProperties;
+import org.apache.tika.parser.ParseContext;
+import org.apache.tika.sax.EmbeddedContentHandler;
+import org.apache.tika.sax.XHTMLContentHandler;
+
+/**
+ * Extracts the icons of a PE file (EXE/DLL) from its resource section and
+ * hands each icon group to the {@link EmbeddedDocumentExtractor} as a
+ * standalone <code>.ico</code> file.
+ * <p>
+ * Windows stores an icon as a group resource ({@code RT_GROUP_ICON}) that
+ * lists the individual images, which are stored as {@code RT_ICON} resources.
+ * The {@code .ico} file format is nearly identical to the group resource;
+ * the only difference is that the group refers to its images by resource id
+ * whereas the file refers to them by file offset. This class rebuilds the
+ * file from the two resource types.
+ * <p>
+ * The resource section is read lazily and only as far as the icon data
+ * reaches: a file without icons costs little more than its resource
+ * directory. File-backed input is read through a positioned channel,
+ * anything else by skipping forward, so non-seekable input works too.
+ * Hostile input is contained by bounds checking every offset, capping the
+ * section size, the number of directory entries visited, the icons per group
+ * and the size of a rebuilt icon, and by refusing cycles in the tree.
+ */
+class PEIconExtractor {
+
+ static final String ICON_MIME_TYPE = "image/vnd.microsoft.icon";
+
+ private static final int RT_ICON = 3;
+ private static final int RT_GROUP_ICON = 14;
+
+ private static final int IMAGE_DIRECTORY_ENTRY_RESOURCE = 2;
+ private static final int PE32_MAGIC = 0x10b;
+ private static final int PE32PLUS_MAGIC = 0x20b;
+ private static final int SECTION_HEADER_SIZE = 40;
+ private static final int RESOURCE_DIRECTORY_SIZE = 16;
+ private static final int RESOURCE_DIRECTORY_ENTRY_SIZE = 8;
+ private static final int RESOURCE_DATA_ENTRY_SIZE = 16;
+ private static final int GRP_ICON_DIR_ENTRY_SIZE = 14;
+ private static final int ICON_DIR_ENTRY_SIZE = 16;
+ private static final long HIGH_BIT = 0x80000000L;
+
+ // Sanity limits for hostile input
+ private static final int MAX_SECTIONS = 96; // the PE spec's own limit
+ private static final int MAX_RESOURCE_SECTION_SIZE = 64 * 1024 * 1024;
+ private static final int MAX_RESOURCE_TREE_DEPTH = 3; // type / name /
language
+ private static final int MAX_DIRECTORY_ENTRIES = 20000; // visited across
the whole tree
+ private static final int MAX_RESOURCE_NAME_LENGTH = 256;
+ private static final int MAX_ICONS_PER_GROUP = 256;
+ // A genuine icon never outgrows the section it came from
+ private static final long MAX_ICO_SIZE = MAX_RESOURCE_SECTION_SIZE;
+
+ private PEIconExtractor() {
+ }
+
+ /**
+ * Continues reading the PE file directly after the COFF file header and
+ * emits every icon group as an embedded document.
+ *
+ * @param stream the input positioned right after the 24 byte COFF
header
+ * @param sizeOptHdrs the SizeOfOptionalHeader field of the COFF header
+ * @param numSections the NumberOfSections field of the COFF header
+ * @param metadata the PE file's own metadata, receives a warning if the
+ * resource tree was too large to walk completely
+ */
+ static void extract(TikaInputStream stream, int sizeOptHdrs, int
numSections,
+ XHTMLContentHandler xhtml, Metadata metadata,
ParseContext context)
+ throws IOException, SAXException, TikaException {
+ if (numSections <= 0 || numSections > MAX_SECTIONS) {
+ return;
+ }
+ // The optional header holds the data directories, of which we
+ // need the one pointing at the resource tree
+ byte[] optHdr = new byte[sizeOptHdrs];
+ IOUtils.readFully(stream, optHdr);
+ int dataDirOffset;
+ switch (sizeOptHdrs >= 2 ? EndianUtils.getUShortLE(optHdr, 0) : 0) {
+ case PE32_MAGIC:
+ dataDirOffset = 96;
+ break;
+ case PE32PLUS_MAGIC:
+ dataDirOffset = 112;
+ break;
+ default:
+ return;
+ }
+ int rsrcEntry = dataDirOffset + IMAGE_DIRECTORY_ENTRY_RESOURCE * 8;
+ if (rsrcEntry + 8 > sizeOptHdrs) {
+ return;
+ }
+ long numDataDirs = EndianUtils.getUIntLE(optHdr, dataDirOffset - 4);
+ if (numDataDirs <= IMAGE_DIRECTORY_ENTRY_RESOURCE) {
+ return;
+ }
+ long rsrcRva = EndianUtils.getUIntLE(optHdr, rsrcEntry);
+ long rsrcSize = EndianUtils.getUIntLE(optHdr, rsrcEntry + 4);
+ if (rsrcRva == 0 || rsrcSize == 0) {
+ return;
+ }
+
+ // The section table tells us where in the file the resource RVA lives
+ byte[] sections = new byte[numSections * SECTION_HEADER_SIZE];
+ IOUtils.readFully(stream, sections);
+ long sectionVa = -1;
+ long sectionRawPtr = -1;
+ long sectionRawSize = -1;
+ for (int i = 0; i < numSections; i++) {
+ int off = i * SECTION_HEADER_SIZE;
+ long va = EndianUtils.getUIntLE(sections, off + 12);
+ long rawSize = EndianUtils.getUIntLE(sections, off + 16);
+ long rawPtr = EndianUtils.getUIntLE(sections, off + 20);
+ if (rsrcRva >= va && rsrcRva < va + rawSize) {
+ sectionVa = va;
+ sectionRawPtr = rawPtr;
+ sectionRawSize = rawSize;
+ break;
+ }
+ }
+ if (sectionVa < 0 || sectionRawSize > MAX_RESOURCE_SECTION_SIZE) {
+ return;
+ }
+
+ InputStream source;
+ if (stream.hasFile()) {
+ FileChannel channel = stream.getFileChannel();
+ if (sectionRawPtr >= channel.size()) {
+ return;
+ }
+ channel.position(sectionRawPtr);
+ source = Channels.newInputStream(channel);
+ } else {
+ // Everything read so far: DOS header up to and including the
section table
+ long position = stream.getPosition();
+ if (sectionRawPtr < position) {
+ return;
+ }
+ IOUtils.skipFully(stream, sectionRawPtr - position);
+ source = stream;
+ }
+
+ // Offsets inside the resource tree are relative to its root, which
+ // normally but not necessarily sits at the start of the section
+ Resources resources = new Resources(new Section(source, (int)
sectionRawSize), sectionVa,
+ rsrcRva - sectionVa);
+ readDirectory(resources, resources.rootOffset, 0, new HashSet<>(), 0,
0, null);
+ if (resources.budget < 0) {
+ EmbeddedDocumentUtil.recordException(new TikaException(
+ "PE resource directory has more than " +
MAX_DIRECTORY_ENTRIES +
+ " entries; icon extraction stopped early"),
metadata, context);
+ }
+ emitIcons(resources, xhtml, context);
+ }
+
+ /**
+ * Walks the three level resource tree (type / name / language) and
+ * collects every icon and icon group. {@code type}, {@code id} and
+ * {@code name} carry what the levels above have established.
+ */
+ private static void readDirectory(Resources resources, long dirOffset, int
depth,
+ Set<Long> visited, int type, int id,
String name)
+ throws IOException {
+ if (depth >= MAX_RESOURCE_TREE_DEPTH || !visited.add(dirOffset)) {
+ return;
+ }
+ Section section = resources.section;
+ int dir = section.index(dirOffset, RESOURCE_DIRECTORY_SIZE);
+ if (dir < 0) {
+ return;
+ }
+ int numEntries = EndianUtils.getUShortLE(section.buf, dir + 12) +
+ EndianUtils.getUShortLE(section.buf, dir + 14);
+ for (int i = 0; i < numEntries; i++) {
+ if (--resources.budget < 0) {
+ return;
+ }
+ int entry = section.index(dirOffset + RESOURCE_DIRECTORY_SIZE +
+ (long) i * RESOURCE_DIRECTORY_ENTRY_SIZE,
RESOURCE_DIRECTORY_ENTRY_SIZE);
+ if (entry < 0) {
+ return;
+ }
+ long nameField = EndianUtils.getUIntLE(section.buf, entry);
+ long dataField = EndianUtils.getUIntLE(section.buf, entry + 4);
+
+ if (depth == 0) {
+ // Only icons are interesting, and a well-formed tree lists
each type once
+ if (nameField != RT_ICON && nameField != RT_GROUP_ICON ||
+ !resources.typesSeen.add((int) nameField)) {
+ continue;
+ }
+ }
+ boolean named = (nameField & HIGH_BIT) != 0;
+ String entryName = null;
+ if (named) {
+ entryName = readName(section, resources.rootOffset +
(nameField & ~HIGH_BIT));
+ if (entryName == null) {
+ continue;
+ }
+ }
+ int entryId = (int) (nameField & 0xffff);
+
+ if ((dataField & HIGH_BIT) != 0) {
+ long subdir = resources.rootOffset + (dataField & ~HIGH_BIT);
+ if (depth == 0) {
+ readDirectory(resources, subdir, 1, visited, entryId, 0,
null);
+ } else {
+ readDirectory(resources, subdir, 2, visited, type,
entryId, entryName);
Review Comment:
The depth cap is bypassed for malformed language entries: once `depth > 0`,
every nested directory is recursed with depth `2`, so a chain below the
language level can grow until the 20,000-entry budget and potentially overflow
the Java stack before the advertised depth limit applies. Increment the depth
on this recursive call so depth-2 directories stop descending.
> Extract icons from PE executables (EXE/DLL) as embedded documents
> -----------------------------------------------------------------
>
> Key: TIKA-4936
> URL: https://issues.apache.org/jira/browse/TIKA-4936
> Project: Tika
> Issue Type: New Feature
> Environment:
>
>
>
> Reporter: Dominik Schmidt
> Priority: Major
>
> h3. Background
> {{ExecutableParser}} currently only reads the COFF file header of PE files
> (EXE/DLL) and emits basic metadata (machine type, architecture bits,
> endianness, created date). The resource section ({{.rsrc}}) is not parsed, so
> resources such as the application icon are not accessible via Tika.
> h3. Proposal
> Parse the PE resource directory and emit each icon group as an embedded
> document:
> * Parse optional header, data directories and section table to locate the
> resource directory (RVA → file offset).
> * Walk the resource tree (type → name/ID → language).
> * For each {{RT_GROUP_ICON}} (type 14), reconstruct a standalone {{.ico}}
> file from the {{GRPICONDIR}} and the referenced {{RT_ICON}} (type 3) entries:
> write an {{ICONDIR}} header and replace the 2-byte resource IDs ({{nID}})
> with 4-byte image offsets.
> * Pass each reconstructed file to the {{EmbeddedDocumentExtractor}} with:
> ** {{Content-Type}}: {{image/vnd.microsoft.icon}}
> ** {{resourceName}}: e.g. {{icon_<id-or-name>.ico}}
> ** {{embeddedResourceType}}: {{THUMBNAIL}} for the first icon group (the icon
> shown by Windows Explorer), {{ATTACHMENT}} for all others
> ** the resource language ID
> Single {{RT_ICON}} entries are intentionally not emitted on their own:
> BMP-based entries are not valid standalone images (no {{BITMAPFILEHEADER}},
> double height for the AND mask), and emitting them would duplicate the group
> data.
> h3. Robustness
> The parser must handle malformed or malicious binaries gracefully:
> * bound the resource tree depth (normally 3 levels) and the number of entries
> * validate all offsets and sizes against the file size
> * guard against cycles in the resource directory
> * failures while extracting resources must not break the existing metadata
> extraction
> h3. Out of scope (possible follow-ups)
> * {{RT_GROUP_CURSOR}}/{{RT_CURSOR}} → {{.cur}}
> * {{RT_MANIFEST}} as an embedded XML document
> * {{VS_VERSIONINFO}} (product name, file version, company) as metadata
> h3. Acceptance criteria
> * Icons of 32- and 64-bit EXE and DLL test files are extracted as valid
> {{.ico}} files that are detected as {{image/vnd.microsoft.icon}}.
> * Icon groups containing both PNG- and BMP-encoded entries are supported.
> * Files without a resource section, or without icons, are parsed as before,
> with no embedded documents.
> * Truncated or corrupted resource sections do not throw and still yield the
> existing metadata.
> * Unit tests use small, license-compatible test files.
--
This message was sent by Atlassian Jira
(v8.20.10#820010)