This is an automated email from the ASF dual-hosted git repository. rzo1 pushed a commit to branch fix/robots-redirect-same-host in repository https://gitbox.apache.org/repos/asf/stormcrawler.git
commit 16854745616dcb599bd30a8fe038355cf75ec92e Author: Richard Zowalla <[email protected]> AuthorDate: Thu Aug 27 14:20:10 2026 +0200 Only follow robots.txt redirects on the same host HttpRobotRulesParser resolved the Location header of a robots.txt redirect and re-fetched it without looking at the scheme, host or port, so the fetch could end up anywhere the redirect pointed to. The redirect is now followed only if the target uses http or https and shares scheme, host and port with the URL it was reached from, otherwise the response is handled like any other one which does not provide rules. This affects sites serving their robots.txt through a redirect to another host, e.g. a CDN: set http.robots.redirect.crosshost.allow to true to keep following those. The existing redirect test follows chains across ports of the same host and sets that option too. --- .../protocol/HttpRobotRulesParser.java | 34 ++++- core/src/main/resources/crawler-default.yaml | 5 + .../HttpRobotRulesParserRedirectTargetTest.java | 158 +++++++++++++++++++++ .../protocol/HttpRobotRulesParserRedirectTest.java | 2 + docs/src/main/asciidoc/configuration.adoc | 1 + 5 files changed, 199 insertions(+), 1 deletion(-) diff --git a/core/src/main/java/org/apache/stormcrawler/protocol/HttpRobotRulesParser.java b/core/src/main/java/org/apache/stormcrawler/protocol/HttpRobotRulesParser.java index 2f846f8b..fc583fc4 100644 --- a/core/src/main/java/org/apache/stormcrawler/protocol/HttpRobotRulesParser.java +++ b/core/src/main/java/org/apache/stormcrawler/protocol/HttpRobotRulesParser.java @@ -44,6 +44,8 @@ public class HttpRobotRulesParser extends RobotRulesParser { protected boolean allow5xx = false; + protected boolean allowCrossHostRedirects = false; + protected Metadata fetchRobotsMd; private static final int MAX_NUM_REDIRECTS = 5; @@ -63,6 +65,8 @@ public class HttpRobotRulesParser extends RobotRulesParser { int robotsTxtContentLimit = ConfUtils.getInt(conf, "http.robots.content.limit", -1); fetchRobotsMd.addValue("http.content.limit", Integer.toString(robotsTxtContentLimit)); allow5xx = ConfUtils.getBoolean(conf, "http.robots.5xx.allow", false); + allowCrossHostRedirects = + ConfUtils.getBoolean(conf, "http.robots.redirect.crosshost.allow", false); } /** Compose unique key to store and access robot rules in cache for given URL. */ @@ -81,6 +85,23 @@ public class HttpRobotRulesParser extends RobotRulesParser { return protocol + ":" + host + ":" + port; } + /** + * Checks whether a redirect while fetching a robots.txt is followed. The target must use the + * http or https scheme and, unless {@code http.robots.redirect.crosshost.allow} is set, share + * scheme, host and port with the URL it was reached from. + * + * @param from URL which returned the redirect + * @param target resolved value of the Location header + * @return true if the target may be fetched + */ + protected boolean followRedirect(URL from, URL target) { + String scheme = target.getProtocol().toLowerCase(Locale.ROOT); + if (!"http".equals(scheme) && !"https".equals(scheme)) { + return false; + } + return allowCrossHostRedirects || getCacheKey(from).equals(getCacheKey(target)); + } + /** * Returns the robots rules from the cache or empty rules if not found. * @@ -149,7 +170,18 @@ public class HttpRobotRulesParser extends RobotRulesParser { String redirection = response.getMetadata().getFirstValue(HttpHeaders.LOCATION); LOG.debug("Redirected from {} to {}", redir, redirection); if (StringUtils.isNotBlank(redirection)) { - redir = URLUtil.resolveUrl(redir, redirection); + URL target = URLUtil.resolveUrl(redir, redirection); + if (!followRedirect(redir, target)) { + LOG.debug( + "Robots for {} redirected to {} which is not fetched " + + "(not on the same host as {})", + url, + target, + redir); + // handled like any other response which does not provide rules + break; + } + redir = target; if (redir.getPath().equals("/robots.txt") && redir.getQuery() == null) { // only if the path (including the query part) of the redirect target is // `/robots.txt` we can get/put the rules from/to the cache under the host diff --git a/core/src/main/resources/crawler-default.yaml b/core/src/main/resources/crawler-default.yaml index 0343f644..fd6b4399 100644 --- a/core/src/main/resources/crawler-default.yaml +++ b/core/src/main/resources/crawler-default.yaml @@ -167,6 +167,11 @@ config: # Allow all if robots.txt cannot be parsed due to a server error (5xx): http.robots.5xx.allow: false + # Follow a robots.txt redirect whose target is on a different scheme, host or + # port than the URL it was reached from? Redirects to schemes other than http + # and https are never followed. + http.robots.redirect.crosshost.allow: false + # ignore directives from robots.txt files? http.robots.file.skip: false diff --git a/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTargetTest.java b/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTargetTest.java new file mode 100644 index 00000000..59245028 --- /dev/null +++ b/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTargetTest.java @@ -0,0 +1,158 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to you under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.stormcrawler.protocol; + +import crawlercommons.robots.BaseRobotRules; +import java.nio.charset.StandardCharsets; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import org.apache.storm.Config; +import org.apache.stormcrawler.Metadata; +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +/** Checks which URLs are fetched when a robots.txt responds with a redirect. */ +class HttpRobotRulesParserRedirectTargetTest { + + private static final String RULES = "User-agent: this_is_only_a_test\nDisallow: /restricted/"; + + /** Protocol stub recording the URLs it is asked to fetch. */ + private static class RecordingProtocol implements Protocol { + + final List<String> requested = new ArrayList<>(); + + final Map<String, ProtocolResponse> responses = new HashMap<>(); + + void redirect(String from, String to) { + Metadata md = new Metadata(); + md.setValue("location", to); + responses.put(from, new ProtocolResponse(new byte[0], 301, md)); + } + + void rules(String url) { + Metadata md = new Metadata(); + md.setValue("content-type", "text/plain"); + responses.put( + url, new ProtocolResponse(RULES.getBytes(StandardCharsets.UTF_8), 200, md)); + } + + @Override + public void configure(Config conf) {} + + @Override + public ProtocolResponse getProtocolOutput(String url, Metadata metadata) { + requested.add(url); + ProtocolResponse response = responses.get(url); + if (response == null) { + return new ProtocolResponse(new byte[0], 404, new Metadata()); + } + return response; + } + + @Override + public BaseRobotRules getRobotRules(String url) { + return null; + } + + @Override + public void cleanup() {} + } + + private static Config conf() { + Config conf = new Config(); + conf.put("http.agent.name", "this_is_only_a_test"); + return conf; + } + + private static HttpRobotRulesParser parser(Config conf) { + HttpRobotRulesParser parser = new HttpRobotRulesParser(); + parser.setConf(conf); + return parser; + } + + @Test + void redirectToSameHostIsFollowed() { + RecordingProtocol protocol = new RecordingProtocol(); + protocol.redirect("http://same.example.org/robots.txt", "/robots/rules.txt"); + protocol.rules("http://same.example.org/robots/rules.txt"); + BaseRobotRules rules = + parser(conf()).getRobotRulesSet(protocol, "http://same.example.org/"); + Assertions.assertTrue( + protocol.requested.contains("http://same.example.org/robots/rules.txt"), + "expected the redirect target to be fetched, requested: " + protocol.requested); + Assertions.assertTrue(rules.isAllowed("http://same.example.org/index.html")); + Assertions.assertFalse(rules.isAllowed("http://same.example.org/restricted/index.html")); + } + + @Test + void redirectToOtherHostIsNotFollowed() { + RecordingProtocol protocol = new RecordingProtocol(); + protocol.redirect( + "http://host.example.org/robots.txt", "http://elsewhere.example.org/robots.txt"); + protocol.rules("http://elsewhere.example.org/robots.txt"); + BaseRobotRules rules = + parser(conf()).getRobotRulesSet(protocol, "http://host.example.org/"); + Assertions.assertFalse( + protocol.requested.contains("http://elsewhere.example.org/robots.txt"), + "robots.txt of another host should not be fetched, requested: " + + protocol.requested); + // no rules obtained, everything is allowed + Assertions.assertTrue(rules.isAllowAll()); + } + + @Test + void redirectToOtherPortIsNotFollowed() { + RecordingProtocol protocol = new RecordingProtocol(); + protocol.redirect( + "http://port.example.org/robots.txt", "http://port.example.org:8080/robots.txt"); + protocol.rules("http://port.example.org:8080/robots.txt"); + BaseRobotRules rules = + parser(conf()).getRobotRulesSet(protocol, "http://port.example.org/"); + Assertions.assertFalse( + protocol.requested.contains("http://port.example.org:8080/robots.txt"), + "robots.txt on another port should not be fetched, requested: " + + protocol.requested); + Assertions.assertTrue(rules.isAllowAll()); + } + + @Test + void redirectToOtherSchemeIsNotFollowed() { + RecordingProtocol protocol = new RecordingProtocol(); + protocol.redirect("http://scheme.example.org/robots.txt", "file:/tmp/robots.txt"); + parser(conf()).getRobotRulesSet(protocol, "http://scheme.example.org/"); + Assertions.assertFalse( + protocol.requested.contains("file:/tmp/robots.txt"), + "robots.txt fetch should stay on http(s), requested: " + protocol.requested); + } + + @Test + void redirectToOtherHostIsFollowedIfConfigured() { + RecordingProtocol protocol = new RecordingProtocol(); + protocol.redirect("http://cdn.example.org/robots.txt", "http://cdn.example.com/robots.txt"); + protocol.rules("http://cdn.example.com/robots.txt"); + Config conf = conf(); + conf.put("http.robots.redirect.crosshost.allow", true); + BaseRobotRules rules = parser(conf).getRobotRulesSet(protocol, "http://cdn.example.org/"); + Assertions.assertTrue( + protocol.requested.contains("http://cdn.example.com/robots.txt"), + "expected the redirect target to be fetched, requested: " + protocol.requested); + Assertions.assertFalse(rules.isAllowed("http://cdn.example.org/restricted/index.html")); + } +} diff --git a/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTest.java b/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTest.java index d3e1eccf..ac6e38da 100644 --- a/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTest.java +++ b/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTest.java @@ -79,6 +79,8 @@ class HttpRobotRulesParserRedirectTest { mockServer7.start(); mockServer8.start(); conf.put("http.agent.name", "this_is_only_a_test"); + // the redirect chains tested below point at other ports of the same host + conf.put("http.robots.redirect.crosshost.allow", true); ProtocolFactory protocolFactory = ProtocolFactory.getInstance(conf); protocol = protocolFactory.getProtocol("http")[0]; protocolFactory.cleanup(); diff --git a/docs/src/main/asciidoc/configuration.adoc b/docs/src/main/asciidoc/configuration.adoc index 7aaa0ccf..04bf76e4 100644 --- a/docs/src/main/asciidoc/configuration.adoc +++ b/docs/src/main/asciidoc/configuration.adoc @@ -220,6 +220,7 @@ implementation. | http.robots.file.skip | false | Ignore robots.txt rules entirely. | http.robots.headers.skip | false | Ignore robots directives from HTTP headers. | http.robots.meta.skip | false | Ignore robots directives from HTML meta tags. +| http.robots.redirect.crosshost.allow | false | Follow a robots.txt redirect pointing at a different scheme, host or port than the URL it was reached from. Redirects to schemes other than http and https are never followed. | http.skip.robots | false | Deprecated (replaced by http.robots.file.skip). | robots.noFollow.strict | true | If true, remove all outlinks from pages marked as noFollow. | http.store.headers | false | Whether to store response headers.
