This is an automated email from the ASF dual-hosted git repository.

rzo1 pushed a commit to branch fix/robots-redirect-same-host
in repository https://gitbox.apache.org/repos/asf/stormcrawler.git

commit 3226dea402f728c449c23859679c8681dd5a1270
Author: Richard Zowalla <[email protected]>
AuthorDate: Thu Aug 27 14:20:10 2026 +0200

    Only follow robots.txt redirects on the same host
    
    HttpRobotRulesParser resolved the Location header of a robots.txt redirect
    and re-fetched it without looking at the scheme, host or port, so the fetch
    could end up anywhere the redirect pointed to. The redirect is now followed
    only if the target uses http or https and shares scheme, host and port with
    the URL it was reached from, otherwise the response is handled like any
    other one which does not provide rules.
    
    This affects sites serving their robots.txt through a redirect to another
    host, e.g. a CDN: set http.robots.redirect.crosshost.allow to true to keep
    following those.
    The existing redirect test follows chains across ports of the same host and
    sets that option too.
---
 .../protocol/HttpRobotRulesParser.java             |  34 ++++-
 core/src/main/resources/crawler-default.yaml       |   5 +
 .../HttpRobotRulesParserRedirectTargetTest.java    | 158 +++++++++++++++++++++
 .../protocol/HttpRobotRulesParserRedirectTest.java |   2 +
 docs/src/main/asciidoc/configuration.adoc          |   1 +
 5 files changed, 199 insertions(+), 1 deletion(-)

diff --git 
a/core/src/main/java/org/apache/stormcrawler/protocol/HttpRobotRulesParser.java 
b/core/src/main/java/org/apache/stormcrawler/protocol/HttpRobotRulesParser.java
index 2f846f8b..fc583fc4 100644
--- 
a/core/src/main/java/org/apache/stormcrawler/protocol/HttpRobotRulesParser.java
+++ 
b/core/src/main/java/org/apache/stormcrawler/protocol/HttpRobotRulesParser.java
@@ -44,6 +44,8 @@ public class HttpRobotRulesParser extends RobotRulesParser {
 
     protected boolean allow5xx = false;
 
+    protected boolean allowCrossHostRedirects = false;
+
     protected Metadata fetchRobotsMd;
 
     private static final int MAX_NUM_REDIRECTS = 5;
@@ -63,6 +65,8 @@ public class HttpRobotRulesParser extends RobotRulesParser {
         int robotsTxtContentLimit = ConfUtils.getInt(conf, 
"http.robots.content.limit", -1);
         fetchRobotsMd.addValue("http.content.limit", 
Integer.toString(robotsTxtContentLimit));
         allow5xx = ConfUtils.getBoolean(conf, "http.robots.5xx.allow", false);
+        allowCrossHostRedirects =
+                ConfUtils.getBoolean(conf, 
"http.robots.redirect.crosshost.allow", false);
     }
 
     /** Compose unique key to store and access robot rules in cache for given 
URL. */
@@ -81,6 +85,23 @@ public class HttpRobotRulesParser extends RobotRulesParser {
         return protocol + ":" + host + ":" + port;
     }
 
+    /**
+     * Checks whether a redirect while fetching a robots.txt is followed. The 
target must use the
+     * http or https scheme and, unless {@code 
http.robots.redirect.crosshost.allow} is set, share
+     * scheme, host and port with the URL it was reached from.
+     *
+     * @param from URL which returned the redirect
+     * @param target resolved value of the Location header
+     * @return true if the target may be fetched
+     */
+    protected boolean followRedirect(URL from, URL target) {
+        String scheme = target.getProtocol().toLowerCase(Locale.ROOT);
+        if (!"http".equals(scheme) && !"https".equals(scheme)) {
+            return false;
+        }
+        return allowCrossHostRedirects || 
getCacheKey(from).equals(getCacheKey(target));
+    }
+
     /**
      * Returns the robots rules from the cache or empty rules if not found.
      *
@@ -149,7 +170,18 @@ public class HttpRobotRulesParser extends RobotRulesParser 
{
                 String redirection = 
response.getMetadata().getFirstValue(HttpHeaders.LOCATION);
                 LOG.debug("Redirected from {} to {}", redir, redirection);
                 if (StringUtils.isNotBlank(redirection)) {
-                    redir = URLUtil.resolveUrl(redir, redirection);
+                    URL target = URLUtil.resolveUrl(redir, redirection);
+                    if (!followRedirect(redir, target)) {
+                        LOG.debug(
+                                "Robots for {} redirected to {} which is not 
fetched "
+                                        + "(not on the same host as {})",
+                                url,
+                                target,
+                                redir);
+                        // handled like any other response which does not 
provide rules
+                        break;
+                    }
+                    redir = target;
                     if (redir.getPath().equals("/robots.txt") && 
redir.getQuery() == null) {
                         // only if the path (including the query part) of the 
redirect target is
                         // `/robots.txt` we can get/put the rules from/to the 
cache under the host
diff --git a/core/src/main/resources/crawler-default.yaml 
b/core/src/main/resources/crawler-default.yaml
index 27092814..fb6f2f35 100644
--- a/core/src/main/resources/crawler-default.yaml
+++ b/core/src/main/resources/crawler-default.yaml
@@ -167,6 +167,11 @@ config:
   # Allow all if robots.txt cannot be parsed due to a server error (5xx):
   http.robots.5xx.allow: false
 
+  # Follow a robots.txt redirect whose target is on a different scheme, host or
+  # port than the URL it was reached from? Redirects to schemes other than http
+  # and https are never followed.
+  http.robots.redirect.crosshost.allow: false
+
   # ignore directives from robots.txt files?
   http.robots.file.skip: false
 
diff --git 
a/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTargetTest.java
 
b/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTargetTest.java
new file mode 100644
index 00000000..59245028
--- /dev/null
+++ 
b/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTargetTest.java
@@ -0,0 +1,158 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to you under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *      http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.stormcrawler.protocol;
+
+import crawlercommons.robots.BaseRobotRules;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+import org.apache.storm.Config;
+import org.apache.stormcrawler.Metadata;
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+/** Checks which URLs are fetched when a robots.txt responds with a redirect. 
*/
+class HttpRobotRulesParserRedirectTargetTest {
+
+    private static final String RULES = "User-agent: 
this_is_only_a_test\nDisallow: /restricted/";
+
+    /** Protocol stub recording the URLs it is asked to fetch. */
+    private static class RecordingProtocol implements Protocol {
+
+        final List<String> requested = new ArrayList<>();
+
+        final Map<String, ProtocolResponse> responses = new HashMap<>();
+
+        void redirect(String from, String to) {
+            Metadata md = new Metadata();
+            md.setValue("location", to);
+            responses.put(from, new ProtocolResponse(new byte[0], 301, md));
+        }
+
+        void rules(String url) {
+            Metadata md = new Metadata();
+            md.setValue("content-type", "text/plain");
+            responses.put(
+                    url, new 
ProtocolResponse(RULES.getBytes(StandardCharsets.UTF_8), 200, md));
+        }
+
+        @Override
+        public void configure(Config conf) {}
+
+        @Override
+        public ProtocolResponse getProtocolOutput(String url, Metadata 
metadata) {
+            requested.add(url);
+            ProtocolResponse response = responses.get(url);
+            if (response == null) {
+                return new ProtocolResponse(new byte[0], 404, new Metadata());
+            }
+            return response;
+        }
+
+        @Override
+        public BaseRobotRules getRobotRules(String url) {
+            return null;
+        }
+
+        @Override
+        public void cleanup() {}
+    }
+
+    private static Config conf() {
+        Config conf = new Config();
+        conf.put("http.agent.name", "this_is_only_a_test");
+        return conf;
+    }
+
+    private static HttpRobotRulesParser parser(Config conf) {
+        HttpRobotRulesParser parser = new HttpRobotRulesParser();
+        parser.setConf(conf);
+        return parser;
+    }
+
+    @Test
+    void redirectToSameHostIsFollowed() {
+        RecordingProtocol protocol = new RecordingProtocol();
+        protocol.redirect("http://same.example.org/robots.txt";, 
"/robots/rules.txt");
+        protocol.rules("http://same.example.org/robots/rules.txt";);
+        BaseRobotRules rules =
+                parser(conf()).getRobotRulesSet(protocol, 
"http://same.example.org/";);
+        Assertions.assertTrue(
+                
protocol.requested.contains("http://same.example.org/robots/rules.txt";),
+                "expected the redirect target to be fetched, requested: " + 
protocol.requested);
+        
Assertions.assertTrue(rules.isAllowed("http://same.example.org/index.html";));
+        
Assertions.assertFalse(rules.isAllowed("http://same.example.org/restricted/index.html";));
+    }
+
+    @Test
+    void redirectToOtherHostIsNotFollowed() {
+        RecordingProtocol protocol = new RecordingProtocol();
+        protocol.redirect(
+                "http://host.example.org/robots.txt";, 
"http://elsewhere.example.org/robots.txt";);
+        protocol.rules("http://elsewhere.example.org/robots.txt";);
+        BaseRobotRules rules =
+                parser(conf()).getRobotRulesSet(protocol, 
"http://host.example.org/";);
+        Assertions.assertFalse(
+                
protocol.requested.contains("http://elsewhere.example.org/robots.txt";),
+                "robots.txt of another host should not be fetched, requested: "
+                        + protocol.requested);
+        // no rules obtained, everything is allowed
+        Assertions.assertTrue(rules.isAllowAll());
+    }
+
+    @Test
+    void redirectToOtherPortIsNotFollowed() {
+        RecordingProtocol protocol = new RecordingProtocol();
+        protocol.redirect(
+                "http://port.example.org/robots.txt";, 
"http://port.example.org:8080/robots.txt";);
+        protocol.rules("http://port.example.org:8080/robots.txt";);
+        BaseRobotRules rules =
+                parser(conf()).getRobotRulesSet(protocol, 
"http://port.example.org/";);
+        Assertions.assertFalse(
+                
protocol.requested.contains("http://port.example.org:8080/robots.txt";),
+                "robots.txt on another port should not be fetched, requested: "
+                        + protocol.requested);
+        Assertions.assertTrue(rules.isAllowAll());
+    }
+
+    @Test
+    void redirectToOtherSchemeIsNotFollowed() {
+        RecordingProtocol protocol = new RecordingProtocol();
+        protocol.redirect("http://scheme.example.org/robots.txt";, 
"file:/tmp/robots.txt");
+        parser(conf()).getRobotRulesSet(protocol, 
"http://scheme.example.org/";);
+        Assertions.assertFalse(
+                protocol.requested.contains("file:/tmp/robots.txt"),
+                "robots.txt fetch should stay on http(s), requested: " + 
protocol.requested);
+    }
+
+    @Test
+    void redirectToOtherHostIsFollowedIfConfigured() {
+        RecordingProtocol protocol = new RecordingProtocol();
+        protocol.redirect("http://cdn.example.org/robots.txt";, 
"http://cdn.example.com/robots.txt";);
+        protocol.rules("http://cdn.example.com/robots.txt";);
+        Config conf = conf();
+        conf.put("http.robots.redirect.crosshost.allow", true);
+        BaseRobotRules rules = parser(conf).getRobotRulesSet(protocol, 
"http://cdn.example.org/";);
+        Assertions.assertTrue(
+                
protocol.requested.contains("http://cdn.example.com/robots.txt";),
+                "expected the redirect target to be fetched, requested: " + 
protocol.requested);
+        
Assertions.assertFalse(rules.isAllowed("http://cdn.example.org/restricted/index.html";));
+    }
+}
diff --git 
a/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTest.java
 
b/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTest.java
index d3e1eccf..ac6e38da 100644
--- 
a/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTest.java
+++ 
b/core/src/test/java/org/apache/stormcrawler/protocol/HttpRobotRulesParserRedirectTest.java
@@ -79,6 +79,8 @@ class HttpRobotRulesParserRedirectTest {
         mockServer7.start();
         mockServer8.start();
         conf.put("http.agent.name", "this_is_only_a_test");
+        // the redirect chains tested below point at other ports of the same 
host
+        conf.put("http.robots.redirect.crosshost.allow", true);
         ProtocolFactory protocolFactory = ProtocolFactory.getInstance(conf);
         protocol = protocolFactory.getProtocol("http")[0];
         protocolFactory.cleanup();
diff --git a/docs/src/main/asciidoc/configuration.adoc 
b/docs/src/main/asciidoc/configuration.adoc
index 40dec13b..688b01b4 100644
--- a/docs/src/main/asciidoc/configuration.adoc
+++ b/docs/src/main/asciidoc/configuration.adoc
@@ -222,6 +222,7 @@ the next major release.
 | http.robots.file.skip | false | Ignore robots.txt rules entirely.
 | http.robots.headers.skip | false | Ignore robots directives from HTTP 
headers.
 | http.robots.meta.skip | false | Ignore robots directives from HTML meta tags.
+| http.robots.redirect.crosshost.allow | false | Follow a robots.txt redirect 
pointing at a different scheme, host or port than the URL it was reached from. 
Redirects to schemes other than http and https are never followed.
 | http.skip.robots | false | Deprecated (replaced by http.robots.file.skip).
 | robots.noFollow.strict | true | If true, remove all outlinks from pages 
marked as noFollow.
 | http.store.headers | false | Whether to store response headers.

Reply via email to