This is an automated email from the ASF dual-hosted git repository.

jnioche pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/stormcrawler.git


The following commit(s) were added to refs/heads/main by this push:
     new 04ee3514 CommaSeparatedToMultivaluedMetadata: replace quadratic token 
append with a single bulk write (#2111)
04ee3514 is described below

commit 04ee3514fc33211b9bb7780d817b01ff15bc65e3
Author: Abhinav <[email protected]>
AuthorDate: Tue Sep 1 22:01:50 2026 +0530

    CommaSeparatedToMultivaluedMetadata: replace quadratic token append with a 
single bulk write (#2111)
    
    * Replace quadratic metadata append in CommaSeparatedToMultivaluedMetadata
    
    Metadata.addValue copies the whole array on every call, so splitting a
    value into n tokens and appending them one at a time in
    CommaSeparatedToMultivaluedMetadata.filter costs n^2/2 element copies.
    The key is removed just before the loop, so the tokens can be stored
    with a single setValues call instead. The number of tokens taken from
    one value is capped (maxTokens, 65536 by default) and a warning is
    logged when the cap trims, so that the amount of metadata a page
    produces does not grow unbounded with the page size.
    
    Metadata.addValues(String, String[]) had the same shape when the key
    was already present: it looped over addValue. It now appends with a
    single array copy; as a side effect it no longer skips blank values in
    that case, which matches what it already did when the key was absent.
    
    * CommaSeparatedToMultivaluedMetadata: cap after blank filtering, default 
128, document maxTokens
    
    - apply the maxTokens cap to the values kept after blank tokens are
      dropped, so that blank tokens do not consume the budget and a low cap
      keeps the first real tokens
    - lower the default cap from 65536 to 128: a page cannot generate an
      unbounded amount of metadata with the default configuration
    - drop the confusing reference to http.content.limit
    - document maxTokens in the parse filters docs and the filter javadoc
    - the scale tests configure maxTokens explicitly since the bulk write
      path they exercise is independent of the cap
---
 .../java/org/apache/stormcrawler/Metadata.java     |  11 +-
 .../CommaSeparatedToMultivaluedMetadata.java       |  40 +++++-
 .../CommaSeparatedToMultivaluedMetadataTest.java   | 150 +++++++++++++++++++++
 docs/src/main/asciidoc/internals.adoc              |   2 +-
 4 files changed, 196 insertions(+), 7 deletions(-)

diff --git a/core/src/main/java/org/apache/stormcrawler/Metadata.java 
b/core/src/main/java/org/apache/stormcrawler/Metadata.java
index 00de145e..62acb221 100644
--- a/core/src/main/java/org/apache/stormcrawler/Metadata.java
+++ b/core/src/main/java/org/apache/stormcrawler/Metadata.java
@@ -213,13 +213,16 @@ public class Metadata {
             return;
         }
         String normalizedKey = normalizeKey(key);
-        if (!md.containsKey(normalizedKey)) {
+        String[] existingvals = md.get(normalizedKey);
+        if (existingvals == null || existingvals.length == 0) {
             md.put(normalizedKey, values);
             return;
         }
-        for (String value : values) {
-            addValue(normalizedKey, value);
-        }
+        // append in one go: adding one value at a time copies the whole array 
for every value
+        String[] newvals = new String[existingvals.length + values.length];
+        System.arraycopy(existingvals, 0, newvals, 0, existingvals.length);
+        System.arraycopy(values, 0, newvals, existingvals.length, 
values.length);
+        md.put(normalizedKey, newvals);
     }
 
     public void addValues(String key, Collection<String> values) {
diff --git 
a/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java
 
b/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java
index 0038b4ab..a1222eab 100644
--- 
a/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java
+++ 
b/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java
@@ -18,23 +18,40 @@
 package org.apache.stormcrawler.parse.filter;
 
 import com.fasterxml.jackson.databind.JsonNode;
+import java.util.ArrayList;
 import java.util.HashSet;
+import java.util.List;
 import java.util.Map;
 import java.util.Set;
+import org.apache.commons.lang3.StringUtils;
 import org.apache.stormcrawler.Metadata;
 import org.apache.stormcrawler.parse.ParseFilter;
 import org.apache.stormcrawler.parse.ParseResult;
 import org.jetbrains.annotations.NotNull;
+import org.slf4j.Logger;
+import org.slf4j.LoggerFactory;
 import org.w3c.dom.DocumentFragment;
 
 /**
  * Rewrites single metadata containing comma separated values into multiple 
values for the same key,
- * useful for instance for keyword tags.
+ * useful for instance for keyword tags. The number of values produced from a 
single entry is
+ * bounded by the optional {@code maxTokens} parameter, 128 by default.
  */
 public class CommaSeparatedToMultivaluedMetadata extends ParseFilter {
 
+    private static final Logger LOG =
+            LoggerFactory.getLogger(CommaSeparatedToMultivaluedMetadata.class);
+
+    /**
+     * Default upper bound on the number of values a single comma separated 
entry can produce, so
+     * that a page cannot generate an unbounded amount of metadata.
+     */
+    private static final int MAX_TOKENS_DEFAULT = 128;
+
     private final Set<String> keys = new HashSet<>();
 
+    private int maxTokens = MAX_TOKENS_DEFAULT;
+
     @Override
     public void configure(@NotNull Map<String, Object> stormConf, @NotNull 
JsonNode filterParams) {
         JsonNode node = filterParams.get("keys");
@@ -48,6 +65,11 @@ public class CommaSeparatedToMultivaluedMetadata extends 
ParseFilter {
         } else {
             keys.add(node.asText());
         }
+
+        node = filterParams.get("maxTokens");
+        if (node != null && node.isInt() && node.asInt() > 0) {
+            maxTokens = node.asInt();
+        }
     }
 
     @Override
@@ -60,9 +82,23 @@ public class CommaSeparatedToMultivaluedMetadata extends 
ParseFilter {
             }
             m.remove(key);
             String[] tokens = val.split(" *, *");
+            // drop blank tokens, as Metadata.addValue used to, then store 
everything in one call:
+            // appending one token at a time copies the whole array for every 
token
+            List<String> values = new ArrayList<>(tokens.length);
             for (String t : tokens) {
-                m.addValue(key, t);
+                if (StringUtils.isNotBlank(t)) {
+                    values.add(t);
+                }
+            }
+            if (values.size() > maxTokens) {
+                LOG.warn(
+                        "Key [{}] has [{}] comma separated values, keeping 
only the first [{}]",
+                        key,
+                        values.size(),
+                        maxTokens);
+                values.subList(maxTokens, values.size()).clear();
             }
+            m.setValues(key, values.toArray(new String[0]));
         }
     }
 }
diff --git 
a/core/src/test/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadataTest.java
 
b/core/src/test/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadataTest.java
new file mode 100644
index 00000000..6c00393c
--- /dev/null
+++ 
b/core/src/test/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadataTest.java
@@ -0,0 +1,150 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to you under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *      http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.stormcrawler.parse.filter;
+
+import com.fasterxml.jackson.databind.JsonNode;
+import com.fasterxml.jackson.databind.ObjectMapper;
+import java.util.Map;
+import org.apache.stormcrawler.Metadata;
+import org.apache.stormcrawler.parse.ParseResult;
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+class CommaSeparatedToMultivaluedMetadataTest {
+
+    /** Counts how often the copy-on-append write path is used. */
+    private static class CountingMetadata extends Metadata {
+        int addValueCalls = 0;
+
+        @Override
+        public void addValue(String key, String value) {
+            addValueCalls++;
+            super.addValue(key, value);
+        }
+    }
+
+    private static CommaSeparatedToMultivaluedMetadata newFilter(String 
jsonParams)
+            throws Exception {
+        ObjectMapper mapper = new ObjectMapper();
+        JsonNode params = mapper.readTree(jsonParams);
+        CommaSeparatedToMultivaluedMetadata filter = new 
CommaSeparatedToMultivaluedMetadata();
+        filter.configure(Map.of(), params);
+        return filter;
+    }
+
+    private static String commaList(int tokens) {
+        StringBuilder sb = new StringBuilder();
+        for (int i = 0; i < tokens; i++) {
+            if (i > 0) {
+                sb.append(',');
+            }
+            sb.append('a');
+        }
+        return sb.toString();
+    }
+
+    @Test
+    void splittingUsesABulkAppend() throws Exception {
+        final String url = "https://example.com/";;
+        CountingMetadata md = new CountingMetadata();
+        md.setValue("parse.keywords", commaList(1000));
+
+        ParseResult parse = new ParseResult();
+        parse.set(url, md);
+
+        // the default cap is deliberately small; the bulk write path is what 
is under test here
+        newFilter("{\"keys\": [\"parse.keywords\"], \"maxTokens\": 1000}")
+                .filter(url, new byte[0], null, parse);
+
+        Assertions.assertEquals(1000, md.getValues("parse.keywords").length);
+        Assertions.assertTrue(
+                md.addValueCalls <= 1,
+                "the filter appended one token at a time: "
+                        + md.addValueCalls
+                        + " calls to Metadata.addValue, each copying the whole 
array");
+    }
+
+    @Test
+    void timeLargeValue() throws Exception {
+        final String url = "https://example.com/";;
+        // 32768 tokens / 65535 chars, far beyond any realistic tag list
+        String value = commaList(32768);
+        Assertions.assertEquals(65535, value.length());
+
+        Metadata md = new Metadata();
+        md.setValue("parse.keywords", value);
+        ParseResult parse = new ParseResult();
+        parse.set(url, md);
+
+        long start = System.nanoTime();
+        // the default cap is deliberately small; the bulk write path is what 
is under test here
+        newFilter("{\"keys\": [\"parse.keywords\"], \"maxTokens\": 65536}")
+                .filter(url, new byte[0], null, parse);
+        long msec = (System.nanoTime() - start) / 1_000_000;
+
+        System.out.println("32768 tokens took " + msec + " ms");
+        Assertions.assertEquals(32768, md.getValues("parse.keywords").length);
+    }
+
+    @Test
+    void splittingHandlesBlankTokens() throws Exception {
+        final String url = "https://example.com/";;
+
+        // empty tokens between consecutive commas are dropped, as 
Metadata.addValue used to
+        Metadata md = new Metadata();
+        md.setValue("parse.keywords", "a,,b");
+        ParseResult parse = new ParseResult();
+        parse.set(url, md);
+        newFilter("{\"keys\": [\"parse.keywords\"]}").filter(url, new byte[0], 
null, parse);
+        Assertions.assertArrayEquals(new String[] {"a", "b"}, 
md.getValues("parse.keywords"));
+
+        // a value made only of empty tokens leaves no entry behind
+        md = new Metadata();
+        md.setValue("parse.keywords", " , ,");
+        parse = new ParseResult();
+        parse.set(url, md);
+        newFilter("{\"keys\": [\"parse.keywords\"]}").filter(url, new byte[0], 
null, parse);
+        Assertions.assertNull(md.getValues("parse.keywords"));
+    }
+
+    @Test
+    void capAppliesAfterBlankTokensAreDropped() throws Exception {
+        final String url = "https://example.com/";;
+
+        // the cap counts real values: blank tokens must not eat into it
+        Metadata md = new Metadata();
+        md.setValue("parse.keywords", ",,,a,b,c");
+        ParseResult parse = new ParseResult();
+        parse.set(url, md);
+        newFilter("{\"keys\": [\"parse.keywords\"], \"maxTokens\": 2}")
+                .filter(url, new byte[0], null, parse);
+        Assertions.assertArrayEquals(new String[] {"a", "b"}, 
md.getValues("parse.keywords"));
+    }
+
+    @Test
+    void defaultCapTrimsLongValues() throws Exception {
+        final String url = "https://example.com/";;
+
+        Metadata md = new Metadata();
+        md.setValue("parse.keywords", commaList(200));
+        ParseResult parse = new ParseResult();
+        parse.set(url, md);
+        newFilter("{\"keys\": [\"parse.keywords\"]}").filter(url, new byte[0], 
null, parse);
+        Assertions.assertEquals(128, md.getValues("parse.keywords").length);
+    }
+}
diff --git a/docs/src/main/asciidoc/internals.adoc 
b/docs/src/main/asciidoc/internals.adoc
index 7c11e40e..09272cc3 100644
--- a/docs/src/main/asciidoc/internals.adoc
+++ b/docs/src/main/asciidoc/internals.adoc
@@ -209,7 +209,7 @@ The archetype includes a default 
link:https://github.com/apache/stormcrawler/blo
 }
 ----
 
-* **CommaSeparatedToMultivaluedMetadata** – 
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java[CommaSeparatedToMultivaluedMetadata]
 rewrites single metadata values containing comma-separated entries into 
multiple values for the same key, useful for keyword tags.
+* **CommaSeparatedToMultivaluedMetadata** – 
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java[CommaSeparatedToMultivaluedMetadata]
 rewrites single metadata values containing comma-separated entries into 
multiple values for the same key, useful for keyword tags. The number of values 
produced from a single entry is bounded by the optional `maxTokens` parameter 
(128 by default), so that a pa [...]
 * **DebugParseFilter** – 
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/DebugParseFilter.java[DebugParseFilter]
 dumps an XML representation of the DOM structure to a temporary file.
 * **DomainParseFilter** – 
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/DomainParseFilter.java[DomainParseFilter]
 stores the domain or host name in the metadata for later indexing.
 * **LDJsonParseFilter** – 
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/LDJsonParseFilter.java[LDJsonParseFilter]
 extracts data from JSON-LD representations.

Reply via email to