This is an automated email from the ASF dual-hosted git repository.

mattcasters pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/hop.git


The following commit(s) were added to refs/heads/main by this push:
     new a704c31532 Allow a custom paragraph separator for the Text Chunker 
transform (#8744)
a704c31532 is described below

commit a704c3153292869a035b5ea7ab8f3b0d4461c9d4
Author: Bart Maertens <[email protected]>
AuthorDate: Mon Oct 5 21:11:51 2026 +0200

    Allow a custom paragraph separator for the Text Chunker transform (#8744)
    
    Adds an optional "paragraph separator" option to the Paragraph chunking
    strategy so paragraphs can be split on a custom, multi-byte literal
    separator (e.g. <PARA>) instead of only blank lines. The separator
    supports variables and the escape sequences \n, \r, \t and \\. When
    empty, the default blank-line splitting (\LF, \CR or \CRLF) is kept.
    
    Includes docs, unit tests and a new integration test with a golden
    data set.
    
    Fixes #8740
---
 .../pages/pipeline/transforms/textchunker.adoc     |   3 +-
 .../0115-text-chunker-paragraph-separator.hpl      | 126 +++++++++++++++++++++
 .../golden-text-chunker-paragraph-separator.csv    |   4 +
 .../transforms/main-0115-text-chunker.hwf          |   3 +
 .../golden-text-chunker-paragraph-separator.json   |  39 +++++++
 ...0115-text-chunker-paragraph-separator UNIT.json |  43 +++++++
 .../pipeline/transforms/chunker/TextChunker.java   |   5 +
 .../transforms/chunker/TextChunkerDialog.java      |  10 +-
 .../transforms/chunker/TextChunkerMeta.java        |  23 ++++
 .../chunking/ParagraphChunkingStrategy.java        |  68 ++++++++++-
 .../chunker/messages/messages_en_US.properties     |   4 +
 .../transforms/chunker/TextChunkerMetaTest.java    |  25 ++++
 .../transforms/chunker/TextChunkerTest.java        |  63 +++++++++++
 .../chunking/ParagraphChunkingStrategyTest.java    |  96 ++++++++++++++++
 14 files changed, 501 insertions(+), 11 deletions(-)

diff --git 
a/docs/hop-user-manual/modules/ROOT/pages/pipeline/transforms/textchunker.adoc 
b/docs/hop-user-manual/modules/ROOT/pages/pipeline/transforms/textchunker.adoc
index 190025be31..460ca7f587 100644
--- 
a/docs/hop-user-manual/modules/ROOT/pages/pipeline/transforms/textchunker.adoc
+++ 
b/docs/hop-user-manual/modules/ROOT/pages/pipeline/transforms/textchunker.adoc
@@ -54,7 +54,7 @@ The transform has no dependency on an AI provider or a model. 
It is plain text p
 |Split on a fixed character count, backing off to the nearest word boundary so 
words are not cut in half. Overlap carries the tail of each chunk into the 
next, which keeps a sentence that straddles a boundary retrievable from both 
sides.
 
 |Paragraph
-|Split on paragraph breaks, packing whole paragraphs up to the chunk size. A 
paragraph longer than the chunk size falls back to character splitting.
+|Split on paragraph breaks, packing whole paragraphs up to the chunk size. A 
paragraph longer than the chunk size falls back to character splitting. By 
default a paragraph break is a blank line (any `\LF`, `\CR` or `\CRLF` line 
ending); the optional *Paragraph separator* option replaces it with a literal 
separator.
 
 |Structure
 |Parse the document into a heading tree and emit one chunk per section, 
prefixing each with a breadcrumb of its heading path so an isolated chunk still 
says where it came from. A section larger than the chunk size falls back to 
character splitting within that section.
@@ -90,6 +90,7 @@ The Hop-native parsers make a pipeline, a workflow or a 
metadata file retrievabl
 |Source document ID field|Optional field holding the identifier of the source 
document. When empty, a row counter is used.
 |Output chunk field|Output field that receives the chunk text.
 |Chunking strategy|Character, Paragraph or Structure, as described above.
+|Paragraph separator|Optional literal separator for the Paragraph strategy, 
for example `<PARA>`. The separator is matched as-is, not as a regular 
expression. When empty, paragraphs are separated by blank lines. Supports 
variables and the escape sequences `` `\n` ``, `` `\r` ``, `` `\t` `` and `` 
`\\` `` so line breaks and tabs can be typed in the field, e.g. `` `\n\n` ``. 
Only used by the Paragraph strategy: the transform reports a warning in the 
pipeline checks when it is set for anoth [...]
 |Content type|Document format used by the Structure strategy.
 |Content type field|Optional field naming the content type per row, for a 
stream that mixes formats.
 |Chunk size|Maximum chunk size in characters. For Paragraph and Structure this 
is a ceiling that triggers the character fallback.
diff --git 
a/integration-tests/transforms/0115-text-chunker-paragraph-separator.hpl 
b/integration-tests/transforms/0115-text-chunker-paragraph-separator.hpl
new file mode 100644
index 0000000000..e880461384
--- /dev/null
+++ b/integration-tests/transforms/0115-text-chunker-paragraph-separator.hpl
@@ -0,0 +1,126 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<!--
+
+Licensed to the Apache Software Foundation (ASF) under one or more
+contributor license agreements.  See the NOTICE file distributed with
+this work for additional information regarding copyright ownership.
+The ASF licenses this file to You under the Apache License, Version 2.0
+(the "License"); you may not use this file except in compliance with
+the License.  You may obtain a copy of the License at
+
+      http://www.apache.org/licenses/LICENSE-2.0
+
+Unless required by applicable law or agreed to in writing, software
+distributed under the License is distributed on an "AS IS" BASIS,
+WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+See the License for the specific language governing permissions and
+limitations under the License.
+
+-->
+<pipeline>
+  <info>
+    <name>0115-text-chunker-paragraph-separator</name>
+    <name_sync_with_filename>Y</name_sync_with_filename>
+    <description>Paragraph chunking with a custom paragraph separator instead 
of blank lines.</description>
+    <extended_description/>
+    <pipeline_version/>
+    <pipeline_type>Normal</pipeline_type>
+    <pipeline_status>0</pipeline_status>
+    <parameters>
+    </parameters>
+    <capture_transform_performance>N</capture_transform_performance>
+    
<transform_performance_capturing_delay>1000</transform_performance_capturing_delay>
+    
<transform_performance_capturing_size_limit>100</transform_performance_capturing_size_limit>
+    <created_user>-</created_user>
+    <created_date>2026/10/03 12:00:00.000</created_date>
+    <modified_user>-</modified_user>
+    <modified_date>2026/10/03 12:00:00.000</modified_date>
+    <key_for_session_key/>
+    <is_key_private>N</is_key_private>
+  </info>
+  <notepads>
+  </notepads>
+  <order>
+    <hop> <from>Document</from> <to>Chunk</to> <enabled>Y</enabled> </hop>
+    <hop> <from>Chunk</from> <to>OUTPUT</to> <enabled>Y</enabled> </hop>
+  </order>
+  <transform>
+    <name>Document</name>
+    <type>DataGrid</type>
+    <description/>
+    <distribute>Y</distribute>
+    <custom_distribution/>
+    <copies>1</copies>
+    <partitioning>
+      <method>none</method>
+      <schema_name/>
+    </partitioning>
+    <data>
+      <line>
+        <item>First part.&lt;PARA&gt;Second part.&lt;PARA&gt;Third part.</item>
+      </line>
+    </data>
+    <fields>
+      <field>
+        <length>-1</length>
+        <precision>-1</precision>
+        <set_empty_string>N</set_empty_string>
+        <name>content</name>
+        <type>String</type>
+      </field>
+    </fields>
+    <attributes/>
+    <GUI>
+      <xloc>96</xloc>
+      <yloc>96</yloc>
+    </GUI>
+  </transform>
+  <transform>
+    <name>Chunk</name>
+    <type>TextChunker</type>
+    <description/>
+    <distribute>Y</distribute>
+    <custom_distribution/>
+    <copies>1</copies>
+    <partitioning>
+      <method>none</method>
+      <schema_name/>
+    </partitioning>
+    <inputField>content</inputField>
+    <outputChunkField>chunk_text</outputChunkField>
+    <chunkingStrategy>PARAGRAPH</chunkingStrategy>
+    <paragraphSeparator>&lt;PARA&gt;</paragraphSeparator>
+    <chunkSize>25</chunkSize>
+    <chunkOverlap>0</chunkOverlap>
+    <includeMetadata>Y</includeMetadata>
+    <chunkIndexField>chunk_index</chunkIndexField>
+    <chunkStartPosField>chunk_start_position</chunkStartPosField>
+    <documentIdField>chunk_doc_id</documentIdField>
+    <chunkCountField>total_chunks</chunkCountField>
+    <attributes/>
+    <GUI>
+      <xloc>320</xloc>
+      <yloc>96</yloc>
+    </GUI>
+  </transform>
+  <transform>
+    <name>OUTPUT</name>
+    <type>Dummy</type>
+    <description/>
+    <distribute>Y</distribute>
+    <custom_distribution/>
+    <copies>1</copies>
+    <partitioning>
+      <method>none</method>
+      <schema_name/>
+    </partitioning>
+    <attributes/>
+    <GUI>
+      <xloc>544</xloc>
+      <yloc>96</yloc>
+    </GUI>
+  </transform>
+  <transform_error_handling>
+  </transform_error_handling>
+  <attributes/>
+</pipeline>
diff --git 
a/integration-tests/transforms/datasets/golden-text-chunker-paragraph-separator.csv
 
b/integration-tests/transforms/datasets/golden-text-chunker-paragraph-separator.csv
new file mode 100644
index 0000000000..5fd8616245
--- /dev/null
+++ 
b/integration-tests/transforms/datasets/golden-text-chunker-paragraph-separator.csv
@@ -0,0 +1,4 @@
+chunk_text,chunk_index,chunk_start_position,total_chunks
+"First part.",0,0,3
+"Second part.",1,17,3
+"Third part.",2,35,3
diff --git a/integration-tests/transforms/main-0115-text-chunker.hwf 
b/integration-tests/transforms/main-0115-text-chunker.hwf
index 9febafd28b..00d468a3e4 100644
--- a/integration-tests/transforms/main-0115-text-chunker.hwf
+++ b/integration-tests/transforms/main-0115-text-chunker.hwf
@@ -60,6 +60,9 @@ limitations under the License.
         <test_name>
           <name>0115-text-chunker-paragraph UNIT</name>
         </test_name>
+        <test_name>
+          <name>0115-text-chunker-paragraph-separator UNIT</name>
+        </test_name>
         <test_name>
           <name>0115-text-chunker-structure-markdown UNIT</name>
         </test_name>
diff --git 
a/integration-tests/transforms/metadata/dataset/golden-text-chunker-paragraph-separator.json
 
b/integration-tests/transforms/metadata/dataset/golden-text-chunker-paragraph-separator.json
new file mode 100644
index 0000000000..927100ebc9
--- /dev/null
+++ 
b/integration-tests/transforms/metadata/dataset/golden-text-chunker-paragraph-separator.json
@@ -0,0 +1,39 @@
+{
+  "base_filename": "golden-text-chunker-paragraph-separator.csv",
+  "name": "golden-text-chunker-paragraph-separator",
+  "description": "Paragraph chunking on a custom <PARA> separator with source 
offsets",
+  "dataset_fields": [
+    {
+      "field_comment": "",
+      "field_length": -1,
+      "field_type": 2,
+      "field_precision": -1,
+      "field_name": "chunk_text",
+      "field_format": ""
+    },
+    {
+      "field_comment": "",
+      "field_length": -1,
+      "field_type": 5,
+      "field_precision": 0,
+      "field_name": "chunk_index",
+      "field_format": "0"
+    },
+    {
+      "field_comment": "",
+      "field_length": -1,
+      "field_type": 5,
+      "field_precision": 0,
+      "field_name": "chunk_start_position",
+      "field_format": "0"
+    },
+    {
+      "field_comment": "",
+      "field_length": -1,
+      "field_type": 5,
+      "field_precision": 0,
+      "field_name": "total_chunks",
+      "field_format": "0"
+    }
+  ]
+}
diff --git 
a/integration-tests/transforms/metadata/unit-test/0115-text-chunker-paragraph-separator
 UNIT.json 
b/integration-tests/transforms/metadata/unit-test/0115-text-chunker-paragraph-separator
 UNIT.json
new file mode 100644
index 0000000000..d06f3e8cbf
--- /dev/null
+++ 
b/integration-tests/transforms/metadata/unit-test/0115-text-chunker-paragraph-separator
 UNIT.json   
@@ -0,0 +1,43 @@
+{
+  "variableValues": [],
+  "database_replacements": [],
+  "autoOpening": false,
+  "basePath": "",
+  "golden_data_sets": [
+    {
+      "field_mappings": [
+        {
+          "transform_field": "chunk_text",
+          "data_set_field": "chunk_text"
+        },
+        {
+          "transform_field": "chunk_index",
+          "data_set_field": "chunk_index"
+        },
+        {
+          "transform_field": "chunk_start_position",
+          "data_set_field": "chunk_start_position"
+        },
+        {
+          "transform_field": "total_chunks",
+          "data_set_field": "total_chunks"
+        }
+      ],
+      "field_order": [
+        "chunk_text",
+        "chunk_index",
+        "chunk_start_position",
+        "total_chunks"
+      ],
+      "transform_name": "OUTPUT",
+      "data_set_name": "golden-text-chunker-paragraph-separator"
+    }
+  ],
+  "input_data_sets": [],
+  "name": "0115-text-chunker-paragraph-separator UNIT",
+  "description": "Paragraph chunking on a custom <PARA> separator with source 
offsets",
+  "trans_test_tweaks": [],
+  "persist_filename": "",
+  "pipeline_filename": "./0115-text-chunker-paragraph-separator.hpl",
+  "test_type": "UNIT_TEST"
+}
diff --git 
a/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunker.java
 
b/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunker.java
index 3a5463d470..0bf89fcae5 100644
--- 
a/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunker.java
+++ 
b/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunker.java
@@ -29,6 +29,7 @@ import org.apache.hop.pipeline.transform.TransformMeta;
 import org.apache.hop.pipeline.transforms.chunker.chunking.ChunkingStrategy;
 import 
org.apache.hop.pipeline.transforms.chunker.chunking.ChunkingStrategyFactory;
 import 
org.apache.hop.pipeline.transforms.chunker.chunking.ChunkingStrategyType;
+import 
org.apache.hop.pipeline.transforms.chunker.chunking.ParagraphChunkingStrategy;
 import 
org.apache.hop.pipeline.transforms.chunker.chunking.StructureChunkingStrategy;
 import org.apache.hop.pipeline.transforms.chunker.document.ContentType;
 import org.apache.hop.pipeline.transforms.chunker.document.ContentTypeResolver;
@@ -78,6 +79,10 @@ public class TextChunker extends 
BaseTransform<TextChunkerMeta, TextChunkerData>
       data.chunkOverlap = 0;
     }
     strategy = 
ChunkingStrategyFactory.createStrategy(meta.getChunkingStrategy());
+    if (strategy instanceof ParagraphChunkingStrategy paragraphStrategy) {
+      paragraphStrategy.setSeparator(
+          
ParagraphChunkingStrategy.decodeEscapes(resolve(meta.getParagraphSeparator())));
+    }
     logBasic(
         BaseMessages.getString(
             PKG, "TextChunker.Log.Initialized", 
String.valueOf(meta.getChunkingStrategy())));
diff --git 
a/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerDialog.java
 
b/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerDialog.java
index cacfc080d9..1943d03041 100644
--- 
a/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerDialog.java
+++ 
b/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerDialog.java
@@ -133,13 +133,15 @@ public class TextChunkerDialog extends 
BaseTransformDialog {
   private void enableFields() {
     // fromString accepts the constant name and the old display text, so this 
follows the widget
     // whatever it holds rather than assuming one spelling.
-    boolean structure =
+    ChunkingStrategyType strategy =
         ChunkingStrategyType.fromString(
-                comboText(
-                    TextChunkerMeta.WIDGET_CHUNKING_STRATEGY, 
input.getChunkingStrategy().name()))
-            == ChunkingStrategyType.STRUCTURE;
+            comboText(
+                TextChunkerMeta.WIDGET_CHUNKING_STRATEGY, 
input.getChunkingStrategy().name()));
+    boolean structure = strategy == ChunkingStrategyType.STRUCTURE;
     setEnabled(TextChunkerMeta.WIDGET_CONTENT_TYPE, structure);
     setEnabled(TextChunkerMeta.WIDGET_CONTENT_TYPE_FIELD, structure);
+    setEnabled(
+        TextChunkerMeta.WIDGET_PARAGRAPH_SEPARATOR, strategy == 
ChunkingStrategyType.PARAGRAPH);
 
     boolean metadata =
         isChecked(TextChunkerMeta.WIDGET_INCLUDE_METADATA, 
input.isIncludeMetadata());
diff --git 
a/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerMeta.java
 
b/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerMeta.java
index 1ae005da8d..addbf66392 100644
--- 
a/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerMeta.java
+++ 
b/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerMeta.java
@@ -66,6 +66,7 @@ public class TextChunkerMeta extends 
BaseTransformMeta<TextChunker, TextChunkerD
   public static final String WIDGET_CONTENT_TYPE = "TEXT_CHUNKER_CONTENT_TYPE";
   public static final String WIDGET_CONTENT_TYPE_FIELD = 
"TEXT_CHUNKER_CONTENT_TYPE_FIELD";
   public static final String WIDGET_CHUNKING_STRATEGY = 
"TEXT_CHUNKER_CHUNKING_STRATEGY";
+  public static final String WIDGET_PARAGRAPH_SEPARATOR = 
"TEXT_CHUNKER_PARAGRAPH_SEPARATOR";
   public static final String WIDGET_INCLUDE_METADATA = 
"TEXT_CHUNKER_INCLUDE_METADATA";
   public static final String WIDGET_CHUNK_INDEX_FIELD = 
"TEXT_CHUNKER_CHUNK_INDEX_FIELD";
   public static final String WIDGET_CHUNK_START_POS_FIELD = 
"TEXT_CHUNKER_CHUNK_START_POS_FIELD";
@@ -114,6 +115,23 @@ public class TextChunkerMeta extends 
BaseTransformMeta<TextChunker, TextChunkerD
   @HopMetadataProperty(key = "chunkingStrategy", injectionKey = 
"CHUNKING_STRATEGY")
   private ChunkingStrategyType chunkingStrategy = 
ChunkingStrategyType.CHARACTER;
 
+  /**
+   * Optional custom paragraph separator for the PARAGRAPH strategy. Empty 
means paragraphs are
+   * separated by blank lines. Supports variables and the escape sequences \n, 
\r, \t and \\.
+   */
+  @GuiWidgetElement(
+      id = WIDGET_PARAGRAPH_SEPARATOR,
+      order = "0450",
+      type = GuiElementType.TEXT,
+      label = "i18n::TextChunker.paragraphSeparator.Label",
+      toolTip = "i18n::TextChunker.paragraphSeparator.Tooltip",
+      variables = true,
+      parentId = GUI_PLUGIN_ELEMENT_PARENT_ID,
+      groupType = GuiWidgetGroupType.BOXES,
+      group = GROUP_CHUNKING)
+  @HopMetadataProperty(key = "paragraphSeparator", injectionKey = 
"PARAGRAPH_SEPARATOR")
+  private String paragraphSeparator = "";
+
   /**
    * The maximum size for each chunk (characters for CHARACTER strategy, 
approximate for PARAGRAPH).
    */
@@ -268,6 +286,7 @@ public class TextChunkerMeta extends 
BaseTransformMeta<TextChunker, TextChunkerD
     inputField = "";
     outputChunkField = "chunk_text";
     chunkingStrategy = ChunkingStrategyType.CHARACTER;
+    paragraphSeparator = "";
     chunkSize = "1000";
     chunkOverlap = "200";
     includeMetadata = true;
@@ -405,6 +424,10 @@ public class TextChunkerMeta extends 
BaseTransformMeta<TextChunker, TextChunkerD
     if (chunkingStrategy == ChunkingStrategyType.PARAGRAPH && resolvedOverlap 
> 0) {
       warning(remarks, transformMeta, 
"TextChunker.Validation.ParagraphOverlapIgnored");
     }
+
+    if (!Utils.isEmpty(paragraphSeparator) && chunkingStrategy != 
ChunkingStrategyType.PARAGRAPH) {
+      warning(remarks, transformMeta, 
"TextChunker.Validation.ParagraphSeparatorIgnored");
+    }
   }
 
   private static void error(
diff --git 
a/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/chunking/ParagraphChunkingStrategy.java
 
b/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/chunking/ParagraphChunkingStrategy.java
index e32d191659..9bb9b949cd 100644
--- 
a/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/chunking/ParagraphChunkingStrategy.java
+++ 
b/plugins/transforms/textchunker/src/main/java/org/apache/hop/pipeline/transforms/chunker/chunking/ParagraphChunkingStrategy.java
@@ -25,9 +25,11 @@ import org.apache.hop.pipeline.transforms.chunker.Chunk;
 /**
  * Chunking strategy that splits text on paragraph boundaries.
  *
- * <p>Paragraphs are identified by blank lines. A paragraph that on its own 
exceeds {@code maxSize}
- * is split further with {@link CharacterChunkingStrategy}, because a chunk 
larger than the limit
- * would be rejected downstream by the embedding model rather than simply 
being large.
+ * <p>Paragraphs are identified by blank lines, or by the literal separator 
configured with {@link
+ * #setSeparator(String)} when the text uses a custom multi-byte marker 
instead. A paragraph that on
+ * its own exceeds {@code maxSize} is split further with {@link 
CharacterChunkingStrategy}, because
+ * a chunk larger than the limit would be rejected downstream by the embedding 
model rather than
+ * simply being large.
  *
  * <p>Consecutive paragraphs are packed into one chunk while they fit within 
{@code maxSize}, which
  * is the usual contract for an embedding chunker: it keeps related text 
together and avoids
@@ -41,8 +43,62 @@ public class ParagraphChunkingStrategy implements 
ChunkingStrategy {
    */
   private static final Pattern PARAGRAPH_SEPARATOR = 
Pattern.compile("\\R\\s*\\R");
 
+  /**
+   * The separator currently in effect. Defaults to blank-line detection and 
is replaced with a
+   * literal custom separator by {@link #setSeparator(String)}.
+   */
+  private Pattern separatorPattern = PARAGRAPH_SEPARATOR;
+
   private final CharacterChunkingStrategy fallback = new 
CharacterChunkingStrategy();
 
+  /**
+   * Sets a custom literal separator to split paragraphs on, so multi-byte 
markers such as {@code
+   * <PARA>} can be used where blank lines do not delimit the text. The 
separator is matched
+   * literally, not as a regular expression. An empty or null separator 
restores the default
+   * blank-line detection.
+   *
+   * @param separator the literal separator, e.g. {@code <PARA>}
+   */
+  public void setSeparator(String separator) {
+    if (separator == null || separator.isEmpty()) {
+      separatorPattern = PARAGRAPH_SEPARATOR;
+    } else {
+      separatorPattern = Pattern.compile(Pattern.quote(separator));
+    }
+  }
+
+  /**
+   * Decodes the escape sequences accepted in a configured separator so line 
breaks can be typed in
+   * a single-line text field: {@code \n}, {@code \r}, {@code \t} and {@code 
\\} (a literal
+   * backslash). Any other sequence is kept as typed, so a separator like 
{@code <PARA>} or {@code
+   * ##} needs no escaping.
+   *
+   * @param separator the separator as typed, may be null
+   * @return the decoded separator, null if the input was null
+   */
+  public static String decodeEscapes(String separator) {
+    if (separator == null || separator.indexOf('\\') < 0) {
+      return separator;
+    }
+    StringBuilder decoded = new StringBuilder(separator.length());
+    for (int i = 0; i < separator.length(); i++) {
+      char c = separator.charAt(i);
+      if (c != '\\' || i + 1 >= separator.length()) {
+        decoded.append(c);
+        continue;
+      }
+      char next = separator.charAt(++i);
+      switch (next) {
+        case 'n' -> decoded.append('\n');
+        case 'r' -> decoded.append('\r');
+        case 't' -> decoded.append('\t');
+        case '\\' -> decoded.append('\\');
+        default -> decoded.append('\\').append(next);
+      }
+    }
+    return decoded.toString();
+  }
+
   @Override
   public List<Chunk> chunk(String text, int maxSize, int overlap) {
     List<Chunk> chunks = new ArrayList<>();
@@ -113,11 +169,11 @@ public class ParagraphChunkingStrategy implements 
ChunkingStrategy {
   }
 
   /**
-   * Splits on blank lines in a single pass, keeping each paragraph's offset 
in the original text.
+   * Splits on the separator in a single pass, keeping each paragraph's offset 
in the original text.
    */
-  private static List<Paragraph> splitParagraphs(String text) {
+  private List<Paragraph> splitParagraphs(String text) {
     List<Paragraph> paragraphs = new ArrayList<>();
-    Matcher matcher = PARAGRAPH_SEPARATOR.matcher(text);
+    Matcher matcher = separatorPattern.matcher(text);
     int start = 0;
     while (matcher.find()) {
       addParagraph(paragraphs, text, start, matcher.start());
diff --git 
a/plugins/transforms/textchunker/src/main/resources/org/apache/hop/pipeline/transforms/chunker/messages/messages_en_US.properties
 
b/plugins/transforms/textchunker/src/main/resources/org/apache/hop/pipeline/transforms/chunker/messages/messages_en_US.properties
index 642b6d0430..ad8baa5616 100644
--- 
a/plugins/transforms/textchunker/src/main/resources/org/apache/hop/pipeline/transforms/chunker/messages/messages_en_US.properties
+++ 
b/plugins/transforms/textchunker/src/main/resources/org/apache/hop/pipeline/transforms/chunker/messages/messages_en_US.properties
@@ -34,6 +34,9 @@ TextChunker.outputChunkField.Tooltip=Name of the field to 
output chunks to
 TextChunker.chunkingStrategy.Label=Chunking strategy
 TextChunker.chunkingStrategy.Tooltip=Choose how to split text: Character, 
Paragraph, or Structure (heading-aware with breadcrumb prefixes)
 
+TextChunker.paragraphSeparator.Label=Paragraph separator
+TextChunker.paragraphSeparator.Tooltip=Optional literal separator between 
paragraphs for the Paragraph strategy, e.g. <PARA>. Leave empty to split on 
blank lines. Supports variables and the escape sequences \\n, \\r, \\t and \\\\.
+
 TextChunker.contentType.Label=Content type
 TextChunker.contentType.Tooltip=Document format for Structure strategy. Auto 
detects from source_type field or text heuristics.
 
@@ -83,3 +86,4 @@ TextChunker.Validation.OutputChunkFieldRequired=Output chunk 
field name must be
 TextChunker.Validation.SourceDocumentIdFieldNotFound=Source document ID field 
''{0}'' not found in the input stream
 TextChunker.Validation.ContentTypeFieldNotFound=Content type field ''{0}'' not 
found in the input stream
 TextChunker.Validation.ParagraphOverlapIgnored=The Paragraph strategy does not 
apply chunk overlap; the configured overlap is ignored
+TextChunker.Validation.ParagraphSeparatorIgnored=The paragraph separator only 
applies to the Paragraph strategy; the configured separator is ignored
diff --git 
a/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerMetaTest.java
 
b/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerMetaTest.java
index bc51b4eab9..3c9e22cece 100644
--- 
a/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerMetaTest.java
+++ 
b/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerMetaTest.java
@@ -64,6 +64,7 @@ class TextChunkerMetaTest {
     original.setChunkCountField("total");
     original.setContentType(ContentType.ASCIIDOC);
     original.setContentTypeField("source_type");
+    original.setParagraphSeparator("<PARA>");
 
     TextChunkerMeta copy = roundTrip(original);
 
@@ -80,6 +81,30 @@ class TextChunkerMetaTest {
     assertEquals(original.getChunkCountField(), copy.getChunkCountField());
     assertEquals(original.getContentType(), copy.getContentType());
     assertEquals(original.getContentTypeField(), copy.getContentTypeField());
+    assertEquals(original.getParagraphSeparator(), 
copy.getParagraphSeparator());
+  }
+
+  /** A separator configured for a non-paragraph strategy is dead config and 
must be flagged. */
+  @Test
+  void checkWarnsWhenTheParagraphSeparatorIsSetForAnotherStrategy() {
+    TextChunkerMeta meta = new TextChunkerMeta();
+    meta.setDefault();
+    meta.setInputField("body");
+    meta.setParagraphSeparator("<PARA>");
+
+    IRowMeta prev = new RowMeta();
+    prev.addValueMeta(new ValueMetaString("body"));
+
+    List<ICheckResult> remarks = new ArrayList<>();
+    meta.check(remarks, null, new TransformMeta(), prev, null, null, null, new 
Variables(), null);
+
+    assertTrue(
+        remarks.stream()
+            .anyMatch(
+                r ->
+                    r.getType() == ICheckResult.TYPE_RESULT_WARNING
+                        && r.getText().toLowerCase().contains("separator")),
+        "a separator set for the Character strategy must produce a warning");
   }
 
   /** Chunk index, start position and total count are numbers, not strings. */
diff --git 
a/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerTest.java
 
b/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerTest.java
index 63d9be9172..5d7edbe830 100644
--- 
a/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerTest.java
+++ 
b/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/TextChunkerTest.java
@@ -227,4 +227,67 @@ class TextChunkerTest {
 
     assertFalse(transform.init(), "a non-positive chunk size must fail init, 
not drop rows");
   }
+
+  /** A multi-byte separator configured on the meta must drive the Paragraph 
strategy end to end. */
+  @Test
+  void splitsOnTheConfiguredParagraphSeparator() throws Exception {
+    TextChunkerMeta meta = new TextChunkerMeta();
+    meta.setDefault();
+    meta.setInputField("text");
+    meta.setChunkingStrategy(ChunkingStrategyType.PARAGRAPH);
+    meta.setParagraphSeparator("<PARA>");
+    meta.setChunkSize("5");
+
+    run(meta, rowMeta(), List.<Object[]>of(new Object[] 
{"one<PARA>two<PARA>three"}));
+
+    int chunkPos = outputRowMeta.indexOfValue("chunk_text");
+    assertEquals(3, output.size());
+    assertEquals("one", output.get(0)[chunkPos]);
+    assertEquals("two", output.get(1)[chunkPos]);
+    assertEquals("three", output.get(2)[chunkPos]);
+  }
+
+  /**
+   * The separator supports variables and escape sequences; a single \n must 
split paragraphs even
+   * though the default blank-line detection would keep the text whole.
+   */
+  @Test
+  void resolvesTheParagraphSeparatorFromVariablesAndEscapes() throws Exception 
{
+    TextChunkerMeta meta = new TextChunkerMeta();
+    meta.setDefault();
+    meta.setInputField("text");
+    meta.setChunkingStrategy(ChunkingStrategyType.PARAGRAPH);
+    meta.setParagraphSeparator("${PARA_SEP}");
+    meta.setChunkSize("8");
+    meta.setChunkOverlap("0");
+
+    TextChunkerData data = new TextChunkerData();
+    TextChunker transform =
+        spy(
+            new TextChunker(
+                helper.transformMeta, meta, data, 0, helper.pipelineMeta, 
helper.pipeline));
+    transform.setVariable("PARA_SEP", "\\n");
+    transform.init();
+    transform.setInputRowMeta(rowMeta());
+
+    Iterator<Object[]> iterator = List.<Object[]>of(new Object[] {"Line 
one\nLine two"}).iterator();
+    doAnswer(invocation -> iterator.hasNext() ? iterator.next() : 
null).when(transform).getRow();
+    doAnswer(
+            invocation -> {
+              outputRowMeta = invocation.getArgument(0);
+              output.add(invocation.getArgument(1));
+              return null;
+            })
+        .when(transform)
+        .putRow(any(IRowMeta.class), any(Object[].class));
+
+    while (transform.processRow()) {
+      // drain
+    }
+
+    int chunkPos = outputRowMeta.indexOfValue("chunk_text");
+    assertEquals(2, output.size(), "a single LF must split when set as the 
separator");
+    assertEquals("Line one", output.get(0)[chunkPos]);
+    assertEquals("Line two", output.get(1)[chunkPos]);
+  }
 }
diff --git 
a/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/chunking/ParagraphChunkingStrategyTest.java
 
b/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/chunking/ParagraphChunkingStrategyTest.java
index a0bea39405..17826d0f51 100644
--- 
a/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/chunking/ParagraphChunkingStrategyTest.java
+++ 
b/plugins/transforms/textchunker/src/test/java/org/apache/hop/pipeline/transforms/chunker/chunking/ParagraphChunkingStrategyTest.java
@@ -18,6 +18,7 @@ package org.apache.hop.pipeline.transforms.chunker.chunking;
 
 import static org.junit.jupiter.api.Assertions.assertEquals;
 import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertNull;
 import static org.junit.jupiter.api.Assertions.assertTrue;
 
 import java.util.List;
@@ -198,6 +199,101 @@ public class ParagraphChunkingStrategyTest {
     assertEquals(ChunkingStrategyType.PARAGRAPH, strategy.getType());
   }
 
+  @Test
+  public void testCustomMultiByteSeparator() {
+    // Blank lines are not the only paragraph marker: some sources mark 
paragraphs with a
+    // multi-byte token.
+    strategy.setSeparator("<PARA>");
+    String text = "First paragraph.<PARA>Second paragraph.<PARA>Third 
paragraph.";
+    List<Chunk> chunks = strategy.chunk(text, 20, 0);
+
+    assertEquals(3, chunks.size());
+    assertEquals("First paragraph.", chunks.get(0).getContent());
+    assertEquals("Second paragraph.", chunks.get(1).getContent());
+    assertEquals("Third paragraph.", chunks.get(2).getContent());
+  }
+
+  @Test
+  public void testCustomSeparatorIsMatchedLiterally() {
+    // A separator is text, not a regular expression: metacharacters must not 
be interpreted.
+    strategy.setSeparator(".*");
+    String text = "alpha.*beta";
+    List<Chunk> chunks = strategy.chunk(text, 5, 0);
+
+    assertEquals(2, chunks.size());
+    assertEquals("alpha", chunks.get(0).getContent());
+    assertEquals("beta", chunks.get(1).getContent());
+  }
+
+  @Test
+  public void testCustomSeparatorKeepsPositionsInTheSourceText() {
+    strategy.setSeparator("<P>");
+    String text = "First.<P>Second.<P>Third.";
+    List<Chunk> chunks = strategy.chunk(text, 7, 0);
+
+    for (Chunk chunk : chunks) {
+      assertEquals(
+          chunk.getContent(),
+          text.substring(chunk.getStartPosition(), chunk.getEndPosition()),
+          "start/end positions must address the chunk in the source text");
+    }
+  }
+
+  @Test
+  public void testBlankLinesDoNotSplitWithACustomSeparator() {
+    strategy.setSeparator("<P>");
+    String text = "Line one\nLine two<P>Line three";
+    List<Chunk> chunks = strategy.chunk(text, 18, 0);
+
+    assertEquals(2, chunks.size());
+    assertEquals("Line one\nLine two", chunks.get(0).getContent());
+  }
+
+  @Test
+  public void testEmptySeparatorRestoresBlankLineSplitting() {
+    strategy.setSeparator("<P>");
+    strategy.setSeparator("");
+    String text = "First paragraph.\n\nSecond paragraph.";
+    List<Chunk> chunks = strategy.chunk(text, 20, 0);
+
+    assertEquals(2, chunks.size());
+  }
+
+  @Test
+  public void testOversizedParagraphStillSplitWithCustomSeparator() {
+    strategy.setSeparator("<P>");
+    String text = "short<P>" + "a".repeat(50);
+    List<Chunk> chunks = strategy.chunk(text, 10, 0);
+
+    assertEquals("short", chunks.get(0).getContent());
+    for (Chunk chunk : chunks) {
+      assertTrue(
+          chunk.getContent().length() <= 10, "chunk exceeds maxSize: '" + 
chunk.getContent() + "'");
+    }
+  }
+
+  @Test
+  public void testPackingStillAppliesWithCustomSeparator() {
+    strategy.setSeparator("<P>");
+    String text = "One.<P>Two.<P>Three.";
+    List<Chunk> chunks = strategy.chunk(text, 1000, 0);
+
+    assertEquals(1, chunks.size());
+    assertEquals(text, chunks.get(0).getContent());
+  }
+
+  @Test
+  public void testDecodeEscapes() {
+    assertEquals("\n\n", ParagraphChunkingStrategy.decodeEscapes("\\n\\n"));
+    assertEquals("\r\n", ParagraphChunkingStrategy.decodeEscapes("\\r\\n"));
+    assertEquals("\t", ParagraphChunkingStrategy.decodeEscapes("\\t"));
+    assertEquals("\\", ParagraphChunkingStrategy.decodeEscapes("\\\\"));
+    // Anything else is separator text, not an escape sequence.
+    assertEquals("<PARA>", ParagraphChunkingStrategy.decodeEscapes("<PARA>"));
+    assertEquals("a\\d", ParagraphChunkingStrategy.decodeEscapes("a\\d"));
+    assertNull(ParagraphChunkingStrategy.decodeEscapes(null));
+  }
+
   @Test
   public void testTextWithTabsAndSpaces() {
     String text = "Paragraph one.\n\n\tParagraph two with tab.\n\n  Paragraph 
three with spaces.";

Reply via email to