This is an automated email from the ASF dual-hosted git repository.
davidzollo pushed a commit to branch dev
in repository https://gitbox.apache.org/repos/asf/seatunnel.git
The following commit(s) were added to refs/heads/dev by this push:
new 7b40dd1acc [Fix][Connector-V2] Harden XML file parsing against XXE
(#11250)
7b40dd1acc is described below
commit 7b40dd1acc8015f45cb5d22aeff906697e06c5c5
Author: Daniel <[email protected]>
AuthorDate: Tue Aug 18 17:37:54 2026 +0800
[Fix][Connector-V2] Harden XML file parsing against XXE (#11250)
Co-authored-by: DanielLeens <[email protected]>
Co-authored-by: David Zollo <[email protected]>
---
docs/en/connectors/source/CosFile.md | 6 ++
docs/en/connectors/source/FtpFile.md | 6 ++
docs/en/connectors/source/HdfsFile.md | 6 ++
docs/en/connectors/source/LocalFile.md | 6 ++
docs/en/connectors/source/OssFile.md | 6 ++
docs/en/connectors/source/OssJindoFile.md | 6 ++
docs/en/connectors/source/S3File.md | 6 ++
docs/en/connectors/source/SftpFile.md | 6 ++
.../introduction/concepts/incompatible-changes.md | 6 ++
docs/zh/connectors/source/CosFile.md | 6 ++
docs/zh/connectors/source/FtpFile.md | 6 ++
docs/zh/connectors/source/HdfsFile.md | 6 ++
docs/zh/connectors/source/LocalFile.md | 6 ++
docs/zh/connectors/source/OssFile.md | 6 ++
docs/zh/connectors/source/OssJindoFile.md | 6 ++
docs/zh/connectors/source/S3File.md | 6 ++
docs/zh/connectors/source/SftpFile.md | 7 ++
.../introduction/concepts/incompatible-changes.md | 6 ++
.../file/source/reader/XmlReadStrategy.java | 49 ++++++++++++-
.../file/source/reader/XmlReadStrategyTest.java | 82 ++++++++++++++++++++++
20 files changed, 239 insertions(+), 1 deletion(-)
diff --git a/docs/en/connectors/source/CosFile.md
b/docs/en/connectors/source/CosFile.md
index 75439fac3f..d2c938bbcb 100644
--- a/docs/en/connectors/source/CosFile.md
+++ b/docs/en/connectors/source/CosFile.md
@@ -353,6 +353,12 @@ Only need to be configured when file_format is xml.
Specifies Whether to process data using the tag attribute format.
+:::caution
+
+For security reasons (XXE hardening), XML files (`file_format_type = xml`)
containing a `<!DOCTYPE ...>` declaration — including benign declarations that
only define internal, non-external entities — are rejected with a
`FILE_READ_FAILED` error. There is no configuration option to restore the
previous, less secure behavior. If your XML files are exported by a tool that
emits a `DOCTYPE` header, remove it or pre-process the file before ingesting it
with SeaTunnel.
+
+:::
+
### csv_use_header_line [boolean]
Whether to use the header line to parse the file, only used when the
file_format is `csv` and the file contains the header line that match RFC 4180
diff --git a/docs/en/connectors/source/FtpFile.md
b/docs/en/connectors/source/FtpFile.md
index 10c94e8498..41bc3e5845 100644
--- a/docs/en/connectors/source/FtpFile.md
+++ b/docs/en/connectors/source/FtpFile.md
@@ -460,6 +460,12 @@ Only need to be configured when file_format is xml.
Specifies Whether to process data using the tag attribute format.
+:::caution
+
+For security reasons (XXE hardening), XML files (`file_format_type = xml`)
containing a `<!DOCTYPE ...>` declaration — including benign declarations that
only define internal, non-external entities — are rejected with a
`FILE_READ_FAILED` error. There is no configuration option to restore the
previous, less secure behavior. If your XML files are exported by a tool that
emits a `DOCTYPE` header, remove it or pre-process the file before ingesting it
with SeaTunnel.
+
+:::
+
### csv_use_header_line [boolean]
Whether to use the header line to parse the file, only used when the
file_format is `csv` and the file contains the header line that match RFC 4180
diff --git a/docs/en/connectors/source/HdfsFile.md
b/docs/en/connectors/source/HdfsFile.md
index ec29badc49..a3cfcc26af 100644
--- a/docs/en/connectors/source/HdfsFile.md
+++ b/docs/en/connectors/source/HdfsFile.md
@@ -152,6 +152,12 @@ The main PDF-specific behaviors are:
Note: Only single-column (top-to-bottom) PDF layouts are supported.
Multi-column layouts (e.g., side-by-side two-column documents) are not
supported and may produce incorrect text ordering.
+:::caution
+
+For security reasons (XXE hardening), XML files (`file_format_type = xml`)
containing a `<!DOCTYPE ...>` declaration — including benign declarations that
only define internal, non-external entities — are rejected with a
`FILE_READ_FAILED` error. There is no configuration option to restore the
previous, less secure behavior. If your XML files are exported by a tool that
emits a `DOCTYPE` header, remove it or pre-process the file before ingesting it
with SeaTunnel.
+
+:::
+
### tables_configs [list]
Use `tables_configs` when one HDFS source needs to read multiple tables or
directories. Each item can define its own
diff --git a/docs/en/connectors/source/LocalFile.md
b/docs/en/connectors/source/LocalFile.md
index 0a327fe104..a8bcefae13 100644
--- a/docs/en/connectors/source/LocalFile.md
+++ b/docs/en/connectors/source/LocalFile.md
@@ -362,6 +362,12 @@ Only need to be configured when file_format is xml.
Specifies Whether to process data using the tag attribute format.
+:::caution
+
+For security reasons (XXE hardening), XML files (`file_format_type = xml`)
containing a `<!DOCTYPE ...>` declaration — including benign declarations that
only define internal, non-external entities — are rejected with a
`FILE_READ_FAILED` error. There is no configuration option to restore the
previous, less secure behavior. If your XML files are exported by a tool that
emits a `DOCTYPE` header, remove it or pre-process the file before ingesting it
with SeaTunnel.
+
+:::
+
### csv_use_header_line [boolean]
Whether to use the header line to parse the file, only used when the
file_format is `csv` and the file contains the header line that match RFC 4180
diff --git a/docs/en/connectors/source/OssFile.md
b/docs/en/connectors/source/OssFile.md
index cab894e741..3b01b7c99d 100644
--- a/docs/en/connectors/source/OssFile.md
+++ b/docs/en/connectors/source/OssFile.md
@@ -280,6 +280,12 @@ The main PDF-specific behaviors are:
Note: Only single-column (top-to-bottom) PDF layouts are supported.
Multi-column layouts (e.g., side-by-side two-column documents) are not
supported and may produce incorrect text ordering.
+:::caution
+
+For security reasons (XXE hardening), XML files (`file_format_type = xml`)
containing a `<!DOCTYPE ...>` declaration — including benign declarations that
only define internal, non-external entities — are rejected with a
`FILE_READ_FAILED` error. There is no configuration option to restore the
previous, less secure behavior. If your XML files are exported by a tool that
emits a `DOCTYPE` header, remove it or pre-process the file before ingesting it
with SeaTunnel.
+
+:::
+
### compress_codec [string]
The compress codec of files and the details that supported as the following
shown:
diff --git a/docs/en/connectors/source/OssJindoFile.md
b/docs/en/connectors/source/OssJindoFile.md
index 1d05a53422..2ce2ebbf23 100644
--- a/docs/en/connectors/source/OssJindoFile.md
+++ b/docs/en/connectors/source/OssJindoFile.md
@@ -227,6 +227,12 @@ The main PDF-specific behaviors are:
Note: Only single-column (top-to-bottom) PDF layouts are supported.
Multi-column layouts (e.g., side-by-side two-column documents) are not
supported and may produce incorrect text ordering.
+:::caution
+
+For security reasons (XXE hardening), XML files (`file_format_type = xml`)
containing a `<!DOCTYPE ...>` declaration — including benign declarations that
only define internal, non-external entities — are rejected with a
`FILE_READ_FAILED` error. There is no configuration option to restore the
previous, less secure behavior. If your XML files are exported by a tool that
emits a `DOCTYPE` header, remove it or pre-process the file before ingesting it
with SeaTunnel.
+
+:::
+
### bucket [string]
The bucket address of oss file system, for example:
`oss://tyrantlucifer-image-bed`
diff --git a/docs/en/connectors/source/S3File.md
b/docs/en/connectors/source/S3File.md
index 9e2b0e3456..1a587a5655 100644
--- a/docs/en/connectors/source/S3File.md
+++ b/docs/en/connectors/source/S3File.md
@@ -290,6 +290,12 @@ The main PDF-specific behaviors are:
Note: Only single-column (top-to-bottom) PDF layouts are supported.
Multi-column layouts (e.g., side-by-side two-column documents) are not
supported and may produce incorrect text ordering.
+:::caution
+
+For security reasons (XXE hardening), XML files (`file_format_type = xml`)
containing a `<!DOCTYPE ...>` declaration — including benign declarations that
only define internal, non-external entities — are rejected with a
`FILE_READ_FAILED` error. There is no configuration option to restore the
previous, less secure behavior. If your XML files are exported by a tool that
emits a `DOCTYPE` header, remove it or pre-process the file before ingesting it
with SeaTunnel.
+
+:::
+
### delimiter/field_delimiter [string]
**delimiter** parameter will deprecate after version 2.3.5, please use
**field_delimiter** instead.
diff --git a/docs/en/connectors/source/SftpFile.md
b/docs/en/connectors/source/SftpFile.md
index f3acf6bdd6..e8e7854271 100644
--- a/docs/en/connectors/source/SftpFile.md
+++ b/docs/en/connectors/source/SftpFile.md
@@ -295,6 +295,12 @@ The main PDF-specific behaviors are:
Note: Only single-column (top-to-bottom) PDF layouts are supported.
Multi-column layouts (e.g., side-by-side two-column documents) are not
supported and may produce incorrect text ordering.
+:::caution
+
+For security reasons (XXE hardening), XML files (`file_format_type = xml`)
containing a `<!DOCTYPE ...>` declaration — including benign declarations that
only define internal, non-external entities — are rejected with a
`FILE_READ_FAILED` error. There is no configuration option to restore the
previous, less secure behavior. If your XML files are exported by a tool that
emits a `DOCTYPE` header, remove it or pre-process the file before ingesting it
with SeaTunnel.
+
+:::
+
### compress_codec [string]
The compress codec of files and the details that supported as the following
shown:
diff --git a/docs/en/introduction/concepts/incompatible-changes.md
b/docs/en/introduction/concepts/incompatible-changes.md
index 8f15b63ce1..a17d39e087 100644
--- a/docs/en/introduction/concepts/incompatible-changes.md
+++ b/docs/en/introduction/concepts/incompatible-changes.md
@@ -128,6 +128,12 @@ You need to check this document before you upgrade to
related version.
- A leftover `flush_interval` key in the `Prometheus` sink block is
rejected only when the config is validated with `--check` / `--dry-run=static`
/ `--dry-run=connect` (which run `validateUnknownKeys`). A directly submitted
job silently ignores the stray key; the connector logs a warning once per sink
writer at startup instead (so a job with parallelism N, multiple tables, or
replicas logs it multiple times).
- **Migration Guide**: Remove `flush_interval` from the `Prometheus` sink
block. To keep timer-based flushing on Zeta, set `sink.flush.interval`
(milliseconds) in the job `env` block. On Spark and Flink, rely on
`batch_size`. The `batch_size` trigger and the final flush on writer close are
unchanged on all engines.
+- **Breaking Change: File connectors reject `DOCTYPE` declarations in XML
input (XXE hardening)**
+ - **Affected component**:
`seatunnel-connectors-v2/connector-file/connector-file-base`
(`XmlReadStrategy`), and every file source built on it: LocalFile, HdfsFile,
S3File, OssFile, OssJindoFile, CosFile, FtpFile, SftpFile (`file_format_type =
xml`)
+ - **Description**: The XML reader previously parsed user-supplied files with
a default dom4j `SAXReader`, leaving DTD processing and external entity
resolution at their JAXP defaults. A crafted `DOCTYPE`/external-entity payload
could disclose local worker-node files, trigger SSRF-style fetches, or exhaust
memory via entity expansion ("billion laughs"). `XmlReadStrategy` now routes
every parse through a hardened reader that enables JAXP secure processing,
rejects any `<!DOCTYPE ...>` de [...]
+ - **Impact**: XML files that previously parsed successfully only because
they carried a `<!DOCTYPE ...>` declaration — even a benign one with no
external `SYSTEM`/`PUBLIC` reference — now fail with
`FileConnectorException(FILE_READ_FAILED)`. There is no configuration option to
opt back into the previous behavior.
+ - **Migration Guide**: Remove the `DOCTYPE` declaration from XML files
before ingesting them with SeaTunnel, or pre-process/re-export the file without
it. Well-formed XML without a `DOCTYPE` declaration is unaffected. (#11250)
+
### Transform Changes
- **[BREAKING]** SQL Transform `PARSEDATETIME`, `TO_DATE`, and `IS_DATE`
functions now only accept whitelisted datetime format patterns. Custom format
patterns that were previously accepted will now fail at runtime. The supported
patterns are:
diff --git a/docs/zh/connectors/source/CosFile.md
b/docs/zh/connectors/source/CosFile.md
index a9e857efa5..e4ed96af59 100644
--- a/docs/zh/connectors/source/CosFile.md
+++ b/docs/zh/connectors/source/CosFile.md
@@ -352,6 +352,12 @@ POI 引擎允许读取的最大 Excel 文件大小,单位为字节。默认值
仅当file_format为xml时才需要配置。
指定是否使用标记属性格式处理数据。
+:::caution
+
+出于安全考虑(XXE 加固), 包含 `<!DOCTYPE ...>` 声明的 XML 文件(`file_format_type =
xml`)——即使是仅定义内部实体、不引用外部资源的良性声明——现在会被拒绝并抛出 `FILE_READ_FAILED`
错误。该行为没有配置项可以恢复为旧版本的处理方式。如果您的 XML 文件由某些工具导出并带有 `DOCTYPE` 头,请在使用 SeaTunnel
读取前将其移除或做预处理。
+
+:::
+
### csv_use_header_line [boolean]
仅在文件格式为 csv 时可以选择配置。
diff --git a/docs/zh/connectors/source/FtpFile.md
b/docs/zh/connectors/source/FtpFile.md
index 3dd50e53f1..26cf5f997d 100644
--- a/docs/zh/connectors/source/FtpFile.md
+++ b/docs/zh/connectors/source/FtpFile.md
@@ -432,6 +432,12 @@ POI 引擎允许读取的最大 Excel 文件大小,单位为字节。默认值
指定是否使用标签属性格式处理数据。
+:::caution
+
+出于安全考虑(XXE 加固), 包含 `<!DOCTYPE ...>` 声明的 XML 文件(`file_format_type =
xml`)——即使是仅定义内部实体、不引用外部资源的良性声明——现在会被拒绝并抛出 `FILE_READ_FAILED`
错误。该行为没有配置项可以恢复为旧版本的处理方式。如果您的 XML 文件由某些工具导出并带有 `DOCTYPE` 头,请在使用 SeaTunnel
读取前将其移除或做预处理。
+
+:::
+
### csv_use_header_line [boolean]
仅在文件格式为 csv 时可以选择配置。
diff --git a/docs/zh/connectors/source/HdfsFile.md
b/docs/zh/connectors/source/HdfsFile.md
index 400ccb5077..af13fed625 100644
--- a/docs/zh/connectors/source/HdfsFile.md
+++ b/docs/zh/connectors/source/HdfsFile.md
@@ -115,6 +115,12 @@ import ChangeLog from
'../changelog/connector-file-hadoop.md';
`text` `csv` `parquet` `orc` `json` `excel` `xml` `binary` `markdown` `pdf`
+:::caution
+
+出于安全考虑(XXE 加固), 包含 `<!DOCTYPE ...>` 声明的 XML 文件(`file_format_type =
xml`)——即使是仅定义内部实体、不引用外部资源的良性声明——现在会被拒绝并抛出 `FILE_READ_FAILED`
错误。该行为没有配置项可以恢复为旧版本的处理方式。如果您的 XML 文件由某些工具导出并带有 `DOCTYPE` 头,请在使用 SeaTunnel
读取前将其移除或做预处理。
+
+:::
+
如果您将文件类型指定为 `markdown`,SeaTunnel 可以解析 markdown 文件并提取结构化数据。
markdown 解析器提取各种元素,包括标题、段落、列表、代码块、表格等。
每个提取出的元素都会转换为一条文档元素结构化记录,schema 如下:
diff --git a/docs/zh/connectors/source/LocalFile.md
b/docs/zh/connectors/source/LocalFile.md
index d318798fe0..3abaffb826 100644
--- a/docs/zh/connectors/source/LocalFile.md
+++ b/docs/zh/connectors/source/LocalFile.md
@@ -362,6 +362,12 @@ POI 引擎允许读取的最大 Excel 文件大小,单位为字节。默认值
指定是否使用标签属性格式处理数据。
+:::caution
+
+出于安全考虑(XXE 加固), 包含 `<!DOCTYPE ...>` 声明的 XML 文件(`file_format_type =
xml`)——即使是仅定义内部实体、不引用外部资源的良性声明——现在会被拒绝并抛出 `FILE_READ_FAILED`
错误。该行为没有配置项可以恢复为旧版本的处理方式。如果您的 XML 文件由某些工具导出并带有 `DOCTYPE` 头,请在使用 SeaTunnel
读取前将其移除或做预处理。
+
+:::
+
### csv_use_header_line [boolean]
是否使用标题行解析文件,仅在 file_format 为 `csv` 且文件包含符合 RFC 4180 的标题行时使用
diff --git a/docs/zh/connectors/source/OssFile.md
b/docs/zh/connectors/source/OssFile.md
index cd8d7f7531..4e1db78541 100644
--- a/docs/zh/connectors/source/OssFile.md
+++ b/docs/zh/connectors/source/OssFile.md
@@ -270,6 +270,12 @@ schema {
`text` `csv` `parquet` `orc` `json` `excel` `xml` `binary` `markdown` `pdf`
+:::caution
+
+出于安全考虑(XXE 加固), 包含 `<!DOCTYPE ...>` 声明的 XML 文件(`file_format_type =
xml`)——即使是仅定义内部实体、不引用外部资源的良性声明——现在会被拒绝并抛出 `FILE_READ_FAILED`
错误。该行为没有配置项可以恢复为旧版本的处理方式。如果您的 XML 文件由某些工具导出并带有 `DOCTYPE` 头,请在使用 SeaTunnel
读取前将其移除或做预处理。
+
+:::
+
如果您将文件类型指定为 `markdown`,SeaTunnel 可以解析 markdown 文件并提取结构化数据。
markdown 解析器提取各种元素,包括标题、段落、列表、代码块、表格等。
每个提取出的元素都会转换为一条文档元素结构化记录,schema 如下:
diff --git a/docs/zh/connectors/source/OssJindoFile.md
b/docs/zh/connectors/source/OssJindoFile.md
index f6763cc518..cf92de98ba 100644
--- a/docs/zh/connectors/source/OssJindoFile.md
+++ b/docs/zh/connectors/source/OssJindoFile.md
@@ -107,6 +107,12 @@ import ChangeLog from
'../changelog/connector-file-oss-jindo.md';
`text` `csv` `parquet` `orc` `json` `excel` `xml` `binary` `markdown`
+:::caution
+
+出于安全考虑(XXE 加固), 包含 `<!DOCTYPE ...>` 声明的 XML 文件(`file_format_type =
xml`)——即使是仅定义内部实体、不引用外部资源的良性声明——现在会被拒绝并抛出 `FILE_READ_FAILED`
错误。该行为没有配置项可以恢复为旧版本的处理方式。如果您的 XML 文件由某些工具导出并带有 `DOCTYPE` 头,请在使用 SeaTunnel
读取前将其移除或做预处理。
+
+:::
+
如果您将文件类型指定为 `markdown`,SeaTunnel 可以解析 markdown 文件并提取结构化数据。
markdown 解析器提取各种元素,包括标题、段落、列表、代码块、表格等。
每个元素都转换为具有以下架构的行:
diff --git a/docs/zh/connectors/source/S3File.md
b/docs/zh/connectors/source/S3File.md
index f3d8a40db8..3a7ca15a24 100644
--- a/docs/zh/connectors/source/S3File.md
+++ b/docs/zh/connectors/source/S3File.md
@@ -403,6 +403,12 @@ abc.*
`text` `csv` `parquet` `orc` `json` `excel` `xml` `binary` `markdown` `pdf`
+:::caution
+
+出于安全考虑(XXE 加固), 包含 `<!DOCTYPE ...>` 声明的 XML 文件(`file_format_type =
xml`)——即使是仅定义内部实体、不引用外部资源的良性声明——现在会被拒绝并抛出 `FILE_READ_FAILED`
错误。该行为没有配置项可以恢复为旧版本的处理方式。如果您的 XML 文件由某些工具导出并带有 `DOCTYPE` 头,请在使用 SeaTunnel
读取前将其移除或做预处理。
+
+:::
+
如果您将文件类型指定为 `markdown`,SeaTunnel 可以解析 markdown 文件并提取结构化数据。
markdown 解析器提取各种元素,包括标题、段落、列表、代码块、表格等。
每个提取出的元素都会转换为一条文档元素结构化记录,schema 如下:
diff --git a/docs/zh/connectors/source/SftpFile.md
b/docs/zh/connectors/source/SftpFile.md
index f8ec822651..f020d621f6 100644
--- a/docs/zh/connectors/source/SftpFile.md
+++ b/docs/zh/connectors/source/SftpFile.md
@@ -191,6 +191,13 @@ abc.*
文件类型,支持以下文件类型:
`text` `csv` `parquet` `orc` `json` `excel` `xml` `binary` `markdown` `pdf`
+
+:::caution
+
+出于安全考虑(XXE 加固), 包含 `<!DOCTYPE ...>` 声明的 XML 文件(`file_format_type =
xml`)——即使是仅定义内部实体、不引用外部资源的良性声明——现在会被拒绝并抛出 `FILE_READ_FAILED`
错误。该行为没有配置项可以恢复为旧版本的处理方式。如果您的 XML 文件由某些工具导出并带有 `DOCTYPE` 头,请在使用 SeaTunnel
读取前将其移除或做预处理。
+
+:::
+
如果您将文件类型指定为`json`,您还应该指定schema选项来告诉连接器如何将数据解析为您想要的行。
例如:
上游数据如下:
diff --git a/docs/zh/introduction/concepts/incompatible-changes.md
b/docs/zh/introduction/concepts/incompatible-changes.md
index a005482e49..359e192edd 100644
--- a/docs/zh/introduction/concepts/incompatible-changes.md
+++ b/docs/zh/introduction/concepts/incompatible-changes.md
@@ -123,6 +123,12 @@
- 只有在使用 `--check` / `--dry-run=static` / `--dry-run=connect` 校验配置时(会执行
`validateUnknownKeys`),`Prometheus` sink 中残留的 `flush_interval`
键才会被拒绝。直接提交的作业会静默忽略该残留键;连接器会在每个 Sink 写入器启动时各打印一次告警作为替代提示(因此并行度为
N、多表或多副本的作业会多次打印)。
- **迁移指南**:从 `Prometheus` sink 中移除 `flush_interval`。如需在 Zeta 上继续使用定时刷新,请在作业
`env` 中设置 `sink.flush.interval`(毫秒)。在 Spark 和 Flink 上请依赖
`batch_size`。`batch_size` 触发和写入器关闭时的最后一次刷新在所有引擎上保持不变。
+- **破坏性变更:File 连接器拒绝 XML 输入中的 `DOCTYPE` 声明(XXE 加固)**
+ -
**影响范围**:`seatunnel-connectors-v2/connector-file/connector-file-base`(`XmlReadStrategy`),以及所有基于该模块构建的
File
Source:LocalFile、HdfsFile、S3File、OssFile、OssJindoFile、CosFile、FtpFile、SftpFile(`file_format_type
= xml`)
+ - **变更说明**:此前 XML 读取器使用默认的 dom4j `SAXReader` 解析用户提供的文件,DTD 处理和外部实体解析均保持 JAXP
默认行为。精心构造的 `DOCTYPE`/外部实体载荷可能导致 worker 节点本地文件泄露、SSRF 式请求,或通过实体展开("billion
laughs")耗尽内存。现在 `XmlReadStrategy` 的所有解析都会经过加固后的 reader:启用 JAXP 安全处理特性、彻底拒绝任何
`<!DOCTYPE ...>` 声明、禁用外部通用/参数实体及外部 DTD 加载,并额外安装一个拒绝一切解析请求的 `EntityResolver`
作为与具体解析器实现无关的兜底防护。
+ - **影响**:此前仅因携带 `<!DOCTYPE ...>` 声明才能被解析的 XML 文件——即使该声明是不引用任何外部
`SYSTEM`/`PUBLIC` 资源的良性声明——现在会以 `FileConnectorException(FILE_READ_FAILED)`
失败。该行为没有配置项可以恢复为旧版本的处理方式。
+ - **迁移指南**:在使用 SeaTunnel 读取前,移除 XML 文件中的 `DOCTYPE` 声明,或对文件做预处理/重新导出。不带
`DOCTYPE` 声明的合法 XML 文件不受影响。(#11250)
+
### 转换变更
- **[BREAKING]** SQL Transform 的 `PARSEDATETIME`、`TO_DATE` 和 `IS_DATE`
函数现在只接受白名单中的日期时间格式模式。以前接受的自定义格式模式现在将在运行时失败。支持的模式有:
diff --git
a/seatunnel-connectors-v2/connector-file/connector-file-base/src/main/java/org/apache/seatunnel/connectors/seatunnel/file/source/reader/XmlReadStrategy.java
b/seatunnel-connectors-v2/connector-file/connector-file-base/src/main/java/org/apache/seatunnel/connectors/seatunnel/file/source/reader/XmlReadStrategy.java
index fcf7daf2dd..823dfd1456 100644
---
a/seatunnel-connectors-v2/connector-file/connector-file-base/src/main/java/org/apache/seatunnel/connectors/seatunnel/file/source/reader/XmlReadStrategy.java
+++
b/seatunnel-connectors-v2/connector-file/connector-file-base/src/main/java/org/apache/seatunnel/connectors/seatunnel/file/source/reader/XmlReadStrategy.java
@@ -48,13 +48,18 @@ import org.dom4j.DocumentException;
import org.dom4j.Element;
import org.dom4j.Node;
import org.dom4j.io.SAXReader;
+import org.xml.sax.InputSource;
+import org.xml.sax.SAXException;
import lombok.SneakyThrows;
import lombok.extern.slf4j.Slf4j;
+import javax.xml.XMLConstants;
+
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStream;
+import java.io.StringReader;
import java.math.BigDecimal;
import java.nio.charset.StandardCharsets;
import java.util.ArrayList;
@@ -69,6 +74,22 @@ import java.util.stream.IntStream;
@Slf4j
public class XmlReadStrategy extends AbstractReadStrategy {
+ /** Reject DTD declarations so XML inputs cannot define
attacker-controlled entities. */
+ private static final String DISALLOW_DOCTYPE_DECL =
+ "http://apache.org/xml/features/disallow-doctype-decl";
+
+ /** Disable external general entities to prevent local file reads and
SSRF. */
+ private static final String EXTERNAL_GENERAL_ENTITIES =
+ "http://xml.org/sax/features/external-general-entities";
+
+ /** Disable external parameter entities to prevent nested external entity
expansion. */
+ private static final String EXTERNAL_PARAMETER_ENTITIES =
+ "http://xml.org/sax/features/external-parameter-entities";
+
+ /** Disable external DTD loading even when a parser implementation
supports it. */
+ private static final String LOAD_EXTERNAL_DTD =
+ "http://apache.org/xml/features/nonvalidating/load-external-dtd";
+
private String tableRowName;
private Boolean useAttrFormat;
private String delimiter;
@@ -104,7 +125,7 @@ public class XmlReadStrategy extends AbstractReadStrategy {
Map<String, String> partitionsMap,
String currentFileName)
throws IOException {
- SAXReader saxReader = new SAXReader();
+ SAXReader saxReader = createSecureSaxReader(split.getFilePath());
Document document;
try (BufferedReader reader = createBomAwareBufferedReader(inputStream,
encoding)) {
document = saxReader.read(reader);
@@ -169,6 +190,32 @@ public class XmlReadStrategy extends AbstractReadStrategy {
});
}
+ /**
+ * Configure the XML reader with XXE-safe defaults before parsing
user-controlled file contents.
+ * The parser-init failure message includes the file path (rather than
reusing the bare "Failed
+ * to initialize secure xml parser" text) so operators can tell an
environment/classpath problem
+ * (every file on this deployment would fail identically) apart from a
per-file issue.
+ */
+ private SAXReader createSecureSaxReader(String filePath) {
+ SAXReader saxReader = new SAXReader();
+ try {
+ // Keep JAXP entity-expansion limits enabled even if Xerces is on
the classpath.
+ saxReader.setFeature(XMLConstants.FEATURE_SECURE_PROCESSING, true);
+ saxReader.setFeature(DISALLOW_DOCTYPE_DECL, true);
+ saxReader.setFeature(EXTERNAL_GENERAL_ENTITIES, false);
+ saxReader.setFeature(EXTERNAL_PARAMETER_ENTITIES, false);
+ saxReader.setFeature(LOAD_EXTERNAL_DTD, false);
+ saxReader.setEntityResolver(
+ (publicId, systemId) -> new InputSource(new
StringReader("")));
+ } catch (SAXException e) {
+ throw new FileConnectorException(
+ FileConnectorErrorCode.FILE_READ_FAILED,
+ "Failed to initialize secure XML parser while reading file
[" + filePath + "]",
+ e);
+ }
+ return saxReader;
+ }
+
@Override
public SeaTunnelRowType getSeaTunnelRowTypeInfo(String path) throws
FileConnectorException {
throw new FileConnectorException(
diff --git
a/seatunnel-connectors-v2/connector-file/connector-file-base/src/test/java/org/apache/seatunnel/connectors/seatunnel/file/source/reader/XmlReadStrategyTest.java
b/seatunnel-connectors-v2/connector-file/connector-file-base/src/test/java/org/apache/seatunnel/connectors/seatunnel/file/source/reader/XmlReadStrategyTest.java
index 01ed161888..10974f7486 100644
---
a/seatunnel-connectors-v2/connector-file/connector-file-base/src/test/java/org/apache/seatunnel/connectors/seatunnel/file/source/reader/XmlReadStrategyTest.java
+++
b/seatunnel-connectors-v2/connector-file/connector-file-base/src/test/java/org/apache/seatunnel/connectors/seatunnel/file/source/reader/XmlReadStrategyTest.java
@@ -27,23 +27,32 @@ import org.apache.seatunnel.api.table.type.SeaTunnelRow;
import org.apache.seatunnel.common.utils.DateTimeUtils;
import org.apache.seatunnel.common.utils.DateUtils;
import org.apache.seatunnel.common.utils.TimeUtils;
+import
org.apache.seatunnel.connectors.seatunnel.file.exception.FileConnectorErrorCode;
+import
org.apache.seatunnel.connectors.seatunnel.file.exception.FileConnectorException;
+import
org.apache.seatunnel.connectors.seatunnel.file.source.split.FileSourceSplit;
import org.apache.seatunnel.connectors.seatunnel.file.util.LocalFileSystemConf;
import org.junit.jupiter.api.Assertions;
import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.io.TempDir;
import lombok.Getter;
+import java.io.ByteArrayInputStream;
import java.io.File;
import java.io.IOException;
import java.math.BigDecimal;
import java.net.URISyntaxException;
import java.net.URL;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
import java.nio.file.Paths;
import java.time.LocalDate;
import java.time.LocalDateTime;
import java.time.LocalTime;
import java.util.ArrayList;
+import java.util.Collections;
import java.util.LinkedHashMap;
import java.util.List;
@@ -121,6 +130,79 @@ public class XmlReadStrategyTest {
}
}
+ @Test
+ public void testXmlReadRejectsExternalEntityPayload(@TempDir Path tempDir)
throws IOException {
+ XmlReadStrategy xmlReadStrategy = createXmlReadStrategy();
+ Path sentinel = tempDir.resolve("seatunnel-xxe.txt");
+ Files.write(sentinel,
Collections.singletonList("secret-from-temp-file"));
+ String xxeXml =
+ "<?xml version=\"1.0\"?>\n"
+ + "<!DOCTYPE row [<!ENTITY xxe SYSTEM \""
+ + sentinel.toUri()
+ + "\">]>\n"
+ + "<RECORDS><RECORD c_string=\"&xxe;\"/></RECORDS>";
+ TestCollector collector = new TestCollector();
+
+ FileConnectorException exception =
+ Assertions.assertThrows(
+ FileConnectorException.class,
+ () ->
+ xmlReadStrategy.readProcess(
+ new FileSourceSplit("xml", "poc.xml"),
+ collector,
+ new ByteArrayInputStream(
+
xxeXml.getBytes(StandardCharsets.UTF_8)),
+ Collections.emptyMap(),
+ "poc.xml"));
+
+ Assertions.assertEquals(
+ FileConnectorErrorCode.FILE_READ_FAILED,
exception.getSeaTunnelErrorCode());
+ Assertions.assertTrue(collector.getRows().isEmpty());
+ Assertions.assertTrue(
+ containsMessage(exception, "DOCTYPE"),
+ "expected secure XML parser to reject the DOCTYPE
declaration");
+ Assertions.assertFalse(
+ containsMessage(exception, "secret-from-temp-file"),
+ "expected the sentinel secret to never appear in the exception
message/cause chain, "
+ + "confirming the external entity was rejected rather
than resolved and merely dropped");
+ }
+
+ private boolean containsMessage(Throwable throwable, String message) {
+ Throwable current = throwable;
+ while (current != null) {
+ if (current.getMessage() != null &&
current.getMessage().contains(message)) {
+ return true;
+ }
+ current = current.getCause();
+ }
+ return false;
+ }
+
+ /** Build a production-like XML reader instance with the shared test
schema loaded. */
+ private XmlReadStrategy createXmlReadStrategy() {
+ Config pluginConfig = loadPluginConfig();
+ XmlReadStrategy xmlReadStrategy = new XmlReadStrategy();
+ LocalFileSystemConf.LocalConf localConf =
+ new LocalFileSystemConf.LocalConf(FS_DEFAULT_NAME_DEFAULT);
+ xmlReadStrategy.setPluginConfig(pluginConfig);
+ xmlReadStrategy.init(localConf);
+ CatalogTable catalogTable =
CatalogTableUtil.buildWithConfig(pluginConfig);
+ xmlReadStrategy.setCatalogTable(catalogTable);
+ return xmlReadStrategy;
+ }
+
+ /** Load the reusable XML test configuration from the module resources. */
+ private Config loadPluginConfig() {
+ URL conf =
XmlReadStrategyTest.class.getResource("/xml/test_read_xml.conf");
+ Assertions.assertNotNull(conf);
+ try {
+ String confPath = Paths.get(conf.toURI()).toString();
+ return ConfigFactory.parseFile(new File(confPath));
+ } catch (URISyntaxException e) {
+ throw new IllegalStateException("Failed to load xml test
configuration", e);
+ }
+ }
+
@Getter
public static class TestCollector implements Collector<SeaTunnelRow> {
private final List<SeaTunnelRow> rows = new ArrayList<>();