This is an automated email from the ASF dual-hosted git repository.
jnioche pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/stormcrawler.git
The following commit(s) were added to refs/heads/main by this push:
new 04ee3514 CommaSeparatedToMultivaluedMetadata: replace quadratic token
append with a single bulk write (#2111)
04ee3514 is described below
commit 04ee3514fc33211b9bb7780d817b01ff15bc65e3
Author: Abhinav <[email protected]>
AuthorDate: Tue Sep 1 22:01:50 2026 +0530
CommaSeparatedToMultivaluedMetadata: replace quadratic token append with a
single bulk write (#2111)
* Replace quadratic metadata append in CommaSeparatedToMultivaluedMetadata
Metadata.addValue copies the whole array on every call, so splitting a
value into n tokens and appending them one at a time in
CommaSeparatedToMultivaluedMetadata.filter costs n^2/2 element copies.
The key is removed just before the loop, so the tokens can be stored
with a single setValues call instead. The number of tokens taken from
one value is capped (maxTokens, 65536 by default) and a warning is
logged when the cap trims, so that the amount of metadata a page
produces does not grow unbounded with the page size.
Metadata.addValues(String, String[]) had the same shape when the key
was already present: it looped over addValue. It now appends with a
single array copy; as a side effect it no longer skips blank values in
that case, which matches what it already did when the key was absent.
* CommaSeparatedToMultivaluedMetadata: cap after blank filtering, default
128, document maxTokens
- apply the maxTokens cap to the values kept after blank tokens are
dropped, so that blank tokens do not consume the budget and a low cap
keeps the first real tokens
- lower the default cap from 65536 to 128: a page cannot generate an
unbounded amount of metadata with the default configuration
- drop the confusing reference to http.content.limit
- document maxTokens in the parse filters docs and the filter javadoc
- the scale tests configure maxTokens explicitly since the bulk write
path they exercise is independent of the cap
---
.../java/org/apache/stormcrawler/Metadata.java | 11 +-
.../CommaSeparatedToMultivaluedMetadata.java | 40 +++++-
.../CommaSeparatedToMultivaluedMetadataTest.java | 150 +++++++++++++++++++++
docs/src/main/asciidoc/internals.adoc | 2 +-
4 files changed, 196 insertions(+), 7 deletions(-)
diff --git a/core/src/main/java/org/apache/stormcrawler/Metadata.java
b/core/src/main/java/org/apache/stormcrawler/Metadata.java
index 00de145e..62acb221 100644
--- a/core/src/main/java/org/apache/stormcrawler/Metadata.java
+++ b/core/src/main/java/org/apache/stormcrawler/Metadata.java
@@ -213,13 +213,16 @@ public class Metadata {
return;
}
String normalizedKey = normalizeKey(key);
- if (!md.containsKey(normalizedKey)) {
+ String[] existingvals = md.get(normalizedKey);
+ if (existingvals == null || existingvals.length == 0) {
md.put(normalizedKey, values);
return;
}
- for (String value : values) {
- addValue(normalizedKey, value);
- }
+ // append in one go: adding one value at a time copies the whole array
for every value
+ String[] newvals = new String[existingvals.length + values.length];
+ System.arraycopy(existingvals, 0, newvals, 0, existingvals.length);
+ System.arraycopy(values, 0, newvals, existingvals.length,
values.length);
+ md.put(normalizedKey, newvals);
}
public void addValues(String key, Collection<String> values) {
diff --git
a/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java
b/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java
index 0038b4ab..a1222eab 100644
---
a/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java
+++
b/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java
@@ -18,23 +18,40 @@
package org.apache.stormcrawler.parse.filter;
import com.fasterxml.jackson.databind.JsonNode;
+import java.util.ArrayList;
import java.util.HashSet;
+import java.util.List;
import java.util.Map;
import java.util.Set;
+import org.apache.commons.lang3.StringUtils;
import org.apache.stormcrawler.Metadata;
import org.apache.stormcrawler.parse.ParseFilter;
import org.apache.stormcrawler.parse.ParseResult;
import org.jetbrains.annotations.NotNull;
+import org.slf4j.Logger;
+import org.slf4j.LoggerFactory;
import org.w3c.dom.DocumentFragment;
/**
* Rewrites single metadata containing comma separated values into multiple
values for the same key,
- * useful for instance for keyword tags.
+ * useful for instance for keyword tags. The number of values produced from a
single entry is
+ * bounded by the optional {@code maxTokens} parameter, 128 by default.
*/
public class CommaSeparatedToMultivaluedMetadata extends ParseFilter {
+ private static final Logger LOG =
+ LoggerFactory.getLogger(CommaSeparatedToMultivaluedMetadata.class);
+
+ /**
+ * Default upper bound on the number of values a single comma separated
entry can produce, so
+ * that a page cannot generate an unbounded amount of metadata.
+ */
+ private static final int MAX_TOKENS_DEFAULT = 128;
+
private final Set<String> keys = new HashSet<>();
+ private int maxTokens = MAX_TOKENS_DEFAULT;
+
@Override
public void configure(@NotNull Map<String, Object> stormConf, @NotNull
JsonNode filterParams) {
JsonNode node = filterParams.get("keys");
@@ -48,6 +65,11 @@ public class CommaSeparatedToMultivaluedMetadata extends
ParseFilter {
} else {
keys.add(node.asText());
}
+
+ node = filterParams.get("maxTokens");
+ if (node != null && node.isInt() && node.asInt() > 0) {
+ maxTokens = node.asInt();
+ }
}
@Override
@@ -60,9 +82,23 @@ public class CommaSeparatedToMultivaluedMetadata extends
ParseFilter {
}
m.remove(key);
String[] tokens = val.split(" *, *");
+ // drop blank tokens, as Metadata.addValue used to, then store
everything in one call:
+ // appending one token at a time copies the whole array for every
token
+ List<String> values = new ArrayList<>(tokens.length);
for (String t : tokens) {
- m.addValue(key, t);
+ if (StringUtils.isNotBlank(t)) {
+ values.add(t);
+ }
+ }
+ if (values.size() > maxTokens) {
+ LOG.warn(
+ "Key [{}] has [{}] comma separated values, keeping
only the first [{}]",
+ key,
+ values.size(),
+ maxTokens);
+ values.subList(maxTokens, values.size()).clear();
}
+ m.setValues(key, values.toArray(new String[0]));
}
}
}
diff --git
a/core/src/test/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadataTest.java
b/core/src/test/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadataTest.java
new file mode 100644
index 00000000..6c00393c
--- /dev/null
+++
b/core/src/test/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadataTest.java
@@ -0,0 +1,150 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to you under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.stormcrawler.parse.filter;
+
+import com.fasterxml.jackson.databind.JsonNode;
+import com.fasterxml.jackson.databind.ObjectMapper;
+import java.util.Map;
+import org.apache.stormcrawler.Metadata;
+import org.apache.stormcrawler.parse.ParseResult;
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+class CommaSeparatedToMultivaluedMetadataTest {
+
+ /** Counts how often the copy-on-append write path is used. */
+ private static class CountingMetadata extends Metadata {
+ int addValueCalls = 0;
+
+ @Override
+ public void addValue(String key, String value) {
+ addValueCalls++;
+ super.addValue(key, value);
+ }
+ }
+
+ private static CommaSeparatedToMultivaluedMetadata newFilter(String
jsonParams)
+ throws Exception {
+ ObjectMapper mapper = new ObjectMapper();
+ JsonNode params = mapper.readTree(jsonParams);
+ CommaSeparatedToMultivaluedMetadata filter = new
CommaSeparatedToMultivaluedMetadata();
+ filter.configure(Map.of(), params);
+ return filter;
+ }
+
+ private static String commaList(int tokens) {
+ StringBuilder sb = new StringBuilder();
+ for (int i = 0; i < tokens; i++) {
+ if (i > 0) {
+ sb.append(',');
+ }
+ sb.append('a');
+ }
+ return sb.toString();
+ }
+
+ @Test
+ void splittingUsesABulkAppend() throws Exception {
+ final String url = "https://example.com/";
+ CountingMetadata md = new CountingMetadata();
+ md.setValue("parse.keywords", commaList(1000));
+
+ ParseResult parse = new ParseResult();
+ parse.set(url, md);
+
+ // the default cap is deliberately small; the bulk write path is what
is under test here
+ newFilter("{\"keys\": [\"parse.keywords\"], \"maxTokens\": 1000}")
+ .filter(url, new byte[0], null, parse);
+
+ Assertions.assertEquals(1000, md.getValues("parse.keywords").length);
+ Assertions.assertTrue(
+ md.addValueCalls <= 1,
+ "the filter appended one token at a time: "
+ + md.addValueCalls
+ + " calls to Metadata.addValue, each copying the whole
array");
+ }
+
+ @Test
+ void timeLargeValue() throws Exception {
+ final String url = "https://example.com/";
+ // 32768 tokens / 65535 chars, far beyond any realistic tag list
+ String value = commaList(32768);
+ Assertions.assertEquals(65535, value.length());
+
+ Metadata md = new Metadata();
+ md.setValue("parse.keywords", value);
+ ParseResult parse = new ParseResult();
+ parse.set(url, md);
+
+ long start = System.nanoTime();
+ // the default cap is deliberately small; the bulk write path is what
is under test here
+ newFilter("{\"keys\": [\"parse.keywords\"], \"maxTokens\": 65536}")
+ .filter(url, new byte[0], null, parse);
+ long msec = (System.nanoTime() - start) / 1_000_000;
+
+ System.out.println("32768 tokens took " + msec + " ms");
+ Assertions.assertEquals(32768, md.getValues("parse.keywords").length);
+ }
+
+ @Test
+ void splittingHandlesBlankTokens() throws Exception {
+ final String url = "https://example.com/";
+
+ // empty tokens between consecutive commas are dropped, as
Metadata.addValue used to
+ Metadata md = new Metadata();
+ md.setValue("parse.keywords", "a,,b");
+ ParseResult parse = new ParseResult();
+ parse.set(url, md);
+ newFilter("{\"keys\": [\"parse.keywords\"]}").filter(url, new byte[0],
null, parse);
+ Assertions.assertArrayEquals(new String[] {"a", "b"},
md.getValues("parse.keywords"));
+
+ // a value made only of empty tokens leaves no entry behind
+ md = new Metadata();
+ md.setValue("parse.keywords", " , ,");
+ parse = new ParseResult();
+ parse.set(url, md);
+ newFilter("{\"keys\": [\"parse.keywords\"]}").filter(url, new byte[0],
null, parse);
+ Assertions.assertNull(md.getValues("parse.keywords"));
+ }
+
+ @Test
+ void capAppliesAfterBlankTokensAreDropped() throws Exception {
+ final String url = "https://example.com/";
+
+ // the cap counts real values: blank tokens must not eat into it
+ Metadata md = new Metadata();
+ md.setValue("parse.keywords", ",,,a,b,c");
+ ParseResult parse = new ParseResult();
+ parse.set(url, md);
+ newFilter("{\"keys\": [\"parse.keywords\"], \"maxTokens\": 2}")
+ .filter(url, new byte[0], null, parse);
+ Assertions.assertArrayEquals(new String[] {"a", "b"},
md.getValues("parse.keywords"));
+ }
+
+ @Test
+ void defaultCapTrimsLongValues() throws Exception {
+ final String url = "https://example.com/";
+
+ Metadata md = new Metadata();
+ md.setValue("parse.keywords", commaList(200));
+ ParseResult parse = new ParseResult();
+ parse.set(url, md);
+ newFilter("{\"keys\": [\"parse.keywords\"]}").filter(url, new byte[0],
null, parse);
+ Assertions.assertEquals(128, md.getValues("parse.keywords").length);
+ }
+}
diff --git a/docs/src/main/asciidoc/internals.adoc
b/docs/src/main/asciidoc/internals.adoc
index 7c11e40e..09272cc3 100644
--- a/docs/src/main/asciidoc/internals.adoc
+++ b/docs/src/main/asciidoc/internals.adoc
@@ -209,7 +209,7 @@ The archetype includes a default
link:https://github.com/apache/stormcrawler/blo
}
----
-* **CommaSeparatedToMultivaluedMetadata** –
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java[CommaSeparatedToMultivaluedMetadata]
rewrites single metadata values containing comma-separated entries into
multiple values for the same key, useful for keyword tags.
+* **CommaSeparatedToMultivaluedMetadata** –
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/CommaSeparatedToMultivaluedMetadata.java[CommaSeparatedToMultivaluedMetadata]
rewrites single metadata values containing comma-separated entries into
multiple values for the same key, useful for keyword tags. The number of values
produced from a single entry is bounded by the optional `maxTokens` parameter
(128 by default), so that a pa [...]
* **DebugParseFilter** –
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/DebugParseFilter.java[DebugParseFilter]
dumps an XML representation of the DOM structure to a temporary file.
* **DomainParseFilter** –
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/DomainParseFilter.java[DomainParseFilter]
stores the domain or host name in the metadata for later indexing.
* **LDJsonParseFilter** –
link:https://github.com/apache/stormcrawler/blob/main/core/src/main/java/org/apache/stormcrawler/parse/filter/LDJsonParseFilter.java[LDJsonParseFilter]
extracts data from JSON-LD representations.