From 30ba69011d8342646414523fac9a03133360f3d4 Mon Sep 17 00:00:00 2001 From: abhinav-phi Date: Sun, 6 Sep 2026 04:59:09 +0530 Subject: [PATCH] Keep the fetcher off private address space by default (#2080) The IP address filter was the only check that looked at the address the fetcher actually connects to, and both of its keys shipped commented out, so the interceptor was never installed at library defaults. The regex exclusion list in the archetypes matched literal localhost and four dotted-quad ranges and nothing else: link-local, CGNAT, IPv6 unique-local and the abbreviated and integer IPv4 forms the resolver maps to 127.0.0.1 all passed it. - crawler-default.yaml ships http.filter.ipaddress.exclude enabled (loopback, RFC1918, link-local, CGNAT, IPv6 unique-local); a fetched page decides which hosts the fetcher connects to, and these ranges host unauthenticated services a public index must not leak into. Crawling an intranet needs an explicit opt-out, documented in the comment next to the key - the archetype default-regex-filters.txt (and its opensearch/solr copies) gains rules for 169.254.0.0/16, 100.64.0.0/10, 0.0.0.0/8, IPv6 fc00::/7 and fe80::/10, the abbreviated loopback forms (127.1) and integer IPv4 hosts (2130706433), and says in its comment that a regex list judges the bytes of the URL only, so the IP filter stays the authoritative check --- .../main/resources/default-regex-filters.txt | 24 +++- core/src/main/resources/crawler-default.yaml | 8 +- .../DefaultRegexFiltersPrivateRangeTest.java | 132 ++++++++++++++++++ .../test/resources/default-regex-filters.txt | 43 +++++- .../main/resources/default-regex-filters.txt | 24 +++- .../main/resources/default-regex-filters.txt | 24 +++- 6 files changed, 245 insertions(+), 10 deletions(-) create mode 100644 core/src/test/java/org/apache/stormcrawler/filtering/DefaultRegexFiltersPrivateRangeTest.java diff --git a/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt b/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt index 389ef587b..36e8dfcb5 100644 --- a/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt +++ b/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt @@ -14,11 +14,23 @@ # of services running on the crawling machine (e.g., Elasticsearch, # Storm) # -# - exclude localhost and loop-back addresses +# NOTE: these rules judge the bytes of the URL only. A host name which +# resolves into one of the ranges below is not caught here; the IP +# address filter (http.filter.ipaddress.exclude) sees the address the +# fetcher actually connects to and is the authoritative check. +# +# - exclude localhost and loop-back addresses, including the abbreviated +# and integer IPv4 forms the resolver maps to 127.0.0.1 # http://localhost:8080 # http://127.0.0.1/ .. http://127.255.255.255/ +# http://127.1/ +# http://2130706433/ # http://[::1]/ --^https?://(?:localhost|127(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){3}|\[::1\])(?::\d+)?(?:/|$) +-^https?://(?:localhost|127(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){0,3}|\[::1\])(?::\d+)?(?:/|$) +# the integer form of an IPv4 address resolves to the address it encodes +# (e.g. http://2130706433/ is 127.0.0.1); such URLs are blocked outright, +# a crawl that needs one can rewrite it to its dotted form +-^https?://\d{1,10}(?::\d+)?(?:/|$) # # - exclude private IP address spaces # 10.0.0.0/8 @@ -27,6 +39,14 @@ -^https?://(?:192\.168(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) # 172.16.0.0/12 -^https?://(?:172\.(?:1[6789]|2[0-9]|3[01])(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 169.254.0.0/16 (link-local, e.g. cloud instance metadata services) +-^https?://(?:169\.254(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 100.64.0.0/10 (carrier-grade NAT) +-^https?://(?:100\.(?:6[0-4]|[7-9][0-9]|1[01][0-9]|12[0-7])(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 0.0.0.0/8 ("this network") +-^https?://(?:0(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){1,3})(?::\d+)?(?:/|$) +# IPv6 unique-local (fc00::/7) and link-local (fe80::/10) addresses +-^https?://\[(?:f[cd][0-9a-f]{2}|fe[89ab][0-9a-f]):[0-9a-f:]+\](?::\d+)?(?:/|$) # accept anything else +. diff --git a/core/src/main/resources/crawler-default.yaml b/core/src/main/resources/crawler-default.yaml index 6945b2c4c..17b49dd9a 100644 --- a/core/src/main/resources/crawler-default.yaml +++ b/core/src/main/resources/crawler-default.yaml @@ -161,8 +161,12 @@ config: # - "linklocal" (matches InetAddress.isLinkLocalAddress()) # Only addresses matching an include rule are fetched (empty means all are # allowed), addresses matching an exclude rule are always blocked. - # http.filter.ipaddress.include: - # http.filter.ipaddress.exclude: "localhost,sitelocal,linklocal" + # The exclude list is enabled by default: a fetched page decides which hosts + # the fetcher connects to, and loopback, private and link-local ranges host + # unauthenticated services (e.g. cloud instance metadata) which a public + # index must not leak into. Set http.filter.ipaddress.include to any value + # (e.g. "0.0.0.0/0") and leave the exclude list empty to crawl an intranet. + http.filter.ipaddress.exclude: "localhost,sitelocal,linklocal,100.64.0.0/10,fd00::/8" # Allow all if robots.txt cannot be parsed due to code 403 (Forbidden): http.robots.403.allow: true diff --git a/core/src/test/java/org/apache/stormcrawler/filtering/DefaultRegexFiltersPrivateRangeTest.java b/core/src/test/java/org/apache/stormcrawler/filtering/DefaultRegexFiltersPrivateRangeTest.java new file mode 100644 index 000000000..3ec60a43b --- /dev/null +++ b/core/src/test/java/org/apache/stormcrawler/filtering/DefaultRegexFiltersPrivateRangeTest.java @@ -0,0 +1,132 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to you under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.stormcrawler.filtering; + +import com.fasterxml.jackson.databind.node.ArrayNode; +import com.fasterxml.jackson.databind.node.JsonNodeFactory; +import com.fasterxml.jackson.databind.node.ObjectNode; +import java.io.IOException; +import java.io.InputStream; +import java.nio.charset.StandardCharsets; +import java.net.MalformedURLException; +import java.net.URL; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import org.apache.stormcrawler.Metadata; +import org.apache.stormcrawler.filtering.regex.RegexURLFilter; +import org.apache.stormcrawler.util.URLUtil; +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.BeforeAll; +import org.junit.jupiter.api.Test; + +/** + * The private-range rules shipped in the archetype default-regex-filters.txt, applied to hosts the + * JVM resolver maps into loopback, link-local and other non-routable space. + */ +class DefaultRegexFiltersPrivateRangeTest { + + private static final List ARCHETYPE_RULES = new ArrayList<>(); + + @BeforeAll + static void loadShippedRules() throws IOException { + try (InputStream in = + DefaultRegexFiltersPrivateRangeTest.class.getResourceAsStream( + "/default-regex-filters.txt")) { + Assertions.assertNotNull(in, "the archetype rules file must be on the test classpath"); + for (String line : + new String(in.readAllBytes(), StandardCharsets.UTF_8).split("\\r?\\n")) { + String trimmed = line.trim(); + if (!trimmed.isEmpty() && !trimmed.startsWith("#")) { + ARCHETYPE_RULES.add(trimmed); + } + } + } + } + + private URLFilter createFilter() { + ObjectNode filterParams = new ObjectNode(JsonNodeFactory.instance); + ArrayNode rules = filterParams.putArray("urlFilters"); + for (String rule : ARCHETYPE_RULES) { + rules.add(rule); + } + RegexURLFilter filter = new RegexURLFilter(); + Map conf = new HashMap<>(); + filter.configure(conf, filterParams); + return filter; + } + + private void assertRejected(URLFilter filter, String url) throws MalformedURLException { + URL source = URLUtil.toURL("http://www.example.com/index.html"); + Assertions.assertNull(filter.filter(source, new Metadata(), url), url); + } + + @Test + void dottedQuadFormsAreRejected() throws MalformedURLException { + URLFilter filter = createFilter(); + assertRejected(filter, "http://127.0.0.1/"); + assertRejected(filter, "http://10.0.0.5/"); + assertRejected(filter, "http://192.168.1.1/"); + assertRejected(filter, "http://172.16.0.1/"); + } + + @Test + void otherNonRoutableRangesAreRejected() throws MalformedURLException { + URLFilter filter = createFilter(); + assertRejected(filter, "http://169.254.169.254/"); + assertRejected(filter, "http://100.64.0.1/"); + assertRejected(filter, "http://100.127.255.254/"); + assertRejected(filter, "http://0.0.0.0/"); + assertRejected(filter, "http://[fd00::1]/"); + assertRejected(filter, "http://[fe80::1]/"); + } + + /** Both host forms are resolved to 127.0.0.1 by InetAddress.getByName. */ + @Test + void abbreviatedAndIntegerLoopbackFormsAreRejected() throws MalformedURLException { + URLFilter filter = createFilter(); + assertRejected(filter, "http://127.1/"); + assertRejected(filter, "http://2130706433/"); + } + + /** The fetcher must not reach private space at all: the shipped IP filter default. */ + @Test + void ipFilterExcludeListIsShippedEnabled() throws Exception { + Map defaults = + org.apache.storm.utils.Utils.findAndReadConfigFile("crawler-default.yaml", false); + Map conf = org.apache.stormcrawler.util.ConfUtils.extractConfigElement(defaults); + String exclude = + org.apache.stormcrawler.util.ConfUtils.getString( + conf, "http.filter.ipaddress.exclude", null); + Assertions.assertNotNull( + exclude, "http.filter.ipaddress.exclude must be enabled in crawler-default.yaml"); + org.apache.stormcrawler.protocol.IPFilterRules ipFilter = + new org.apache.stormcrawler.protocol.IPFilterRules(conf); + Assertions.assertFalse(ipFilter.isEmpty()); + Assertions.assertFalse( + ipFilter.accept(java.net.InetAddress.getByName("169.254.169.254")), + "link-local must be excluded by the shipped default"); + Assertions.assertFalse( + ipFilter.accept(java.net.InetAddress.getByName("127.0.0.1")), + "loopback must be excluded by the shipped default"); + Assertions.assertTrue( + ipFilter.accept(java.net.InetAddress.getByName("140.211.11.131")), + "public addresses must still be accepted"); + } +} diff --git a/core/src/test/resources/default-regex-filters.txt b/core/src/test/resources/default-regex-filters.txt index cfd2979e7..36e8dfcb5 100644 --- a/core/src/test/resources/default-regex-filters.txt +++ b/core/src/test/resources/default-regex-filters.txt @@ -3,11 +3,50 @@ # skip image and other suffixes we can't parse or are not likely to be relevant # if you want to crawl images or videos or archives then you should comment out this line --(?i)\.(gif|jpg|png|ico|css|sit|eps|wmf|zip|gz|rpm|tgz|mov|exe|jpeg|bmp|js|mpg|mp3|mp4)(\?|&|$) +-(?i)\.(apk|deb|cab|iso|gif|jpg|png|svg|ico|css|sit|eps|wmf|rar|tar|jar|zip|gz|bz2|rpm|tgz|mov|exe|jpeg|jpe|bmp|js|mpg|mp3|mp4|m4a|ogv|kml|wmv|swf|flv|mkv|m4v|webm|ra|wma|wav|avi|xspf|m3u)(\?|&|$) # skip URLs with slash-delimited segment that repeats 3+ times, to break loops -# very time-consuming : use only if necessary +# very time-consuming : use BasicURLFilter instead # -.*(/[^/]+)/[^/]+\1/[^/]+\1/ +# exclude localhost and equivalents to avoid that information +# can be leaked by placing faked links pointing to web interfaces +# of services running on the crawling machine (e.g., Elasticsearch, +# Storm) +# +# NOTE: these rules judge the bytes of the URL only. A host name which +# resolves into one of the ranges below is not caught here; the IP +# address filter (http.filter.ipaddress.exclude) sees the address the +# fetcher actually connects to and is the authoritative check. +# +# - exclude localhost and loop-back addresses, including the abbreviated +# and integer IPv4 forms the resolver maps to 127.0.0.1 +# http://localhost:8080 +# http://127.0.0.1/ .. http://127.255.255.255/ +# http://127.1/ +# http://2130706433/ +# http://[::1]/ +-^https?://(?:localhost|127(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){0,3}|\[::1\])(?::\d+)?(?:/|$) +# the integer form of an IPv4 address resolves to the address it encodes +# (e.g. http://2130706433/ is 127.0.0.1); such URLs are blocked outright, +# a crawl that needs one can rewrite it to its dotted form +-^https?://\d{1,10}(?::\d+)?(?:/|$) +# +# - exclude private IP address spaces +# 10.0.0.0/8 +-^https?://(?:10(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){3})(?::\d+)?(?:/|$) +# 192.168.0.0/16 +-^https?://(?:192\.168(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 172.16.0.0/12 +-^https?://(?:172\.(?:1[6789]|2[0-9]|3[01])(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 169.254.0.0/16 (link-local, e.g. cloud instance metadata services) +-^https?://(?:169\.254(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 100.64.0.0/10 (carrier-grade NAT) +-^https?://(?:100\.(?:6[0-4]|[7-9][0-9]|1[01][0-9]|12[0-7])(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 0.0.0.0/8 ("this network") +-^https?://(?:0(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){1,3})(?::\d+)?(?:/|$) +# IPv6 unique-local (fc00::/7) and link-local (fe80::/10) addresses +-^https?://\[(?:f[cd][0-9a-f]{2}|fe[89ab][0-9a-f]):[0-9a-f:]+\](?::\d+)?(?:/|$) + # accept anything else +. diff --git a/external/opensearch/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt b/external/opensearch/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt index 389ef587b..36e8dfcb5 100644 --- a/external/opensearch/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt +++ b/external/opensearch/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt @@ -14,11 +14,23 @@ # of services running on the crawling machine (e.g., Elasticsearch, # Storm) # -# - exclude localhost and loop-back addresses +# NOTE: these rules judge the bytes of the URL only. A host name which +# resolves into one of the ranges below is not caught here; the IP +# address filter (http.filter.ipaddress.exclude) sees the address the +# fetcher actually connects to and is the authoritative check. +# +# - exclude localhost and loop-back addresses, including the abbreviated +# and integer IPv4 forms the resolver maps to 127.0.0.1 # http://localhost:8080 # http://127.0.0.1/ .. http://127.255.255.255/ +# http://127.1/ +# http://2130706433/ # http://[::1]/ --^https?://(?:localhost|127(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){3}|\[::1\])(?::\d+)?(?:/|$) +-^https?://(?:localhost|127(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){0,3}|\[::1\])(?::\d+)?(?:/|$) +# the integer form of an IPv4 address resolves to the address it encodes +# (e.g. http://2130706433/ is 127.0.0.1); such URLs are blocked outright, +# a crawl that needs one can rewrite it to its dotted form +-^https?://\d{1,10}(?::\d+)?(?:/|$) # # - exclude private IP address spaces # 10.0.0.0/8 @@ -27,6 +39,14 @@ -^https?://(?:192\.168(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) # 172.16.0.0/12 -^https?://(?:172\.(?:1[6789]|2[0-9]|3[01])(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 169.254.0.0/16 (link-local, e.g. cloud instance metadata services) +-^https?://(?:169\.254(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 100.64.0.0/10 (carrier-grade NAT) +-^https?://(?:100\.(?:6[0-4]|[7-9][0-9]|1[01][0-9]|12[0-7])(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 0.0.0.0/8 ("this network") +-^https?://(?:0(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){1,3})(?::\d+)?(?:/|$) +# IPv6 unique-local (fc00::/7) and link-local (fe80::/10) addresses +-^https?://\[(?:f[cd][0-9a-f]{2}|fe[89ab][0-9a-f]):[0-9a-f:]+\](?::\d+)?(?:/|$) # accept anything else +. diff --git a/external/solr/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt b/external/solr/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt index 389ef587b..36e8dfcb5 100644 --- a/external/solr/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt +++ b/external/solr/archetype/src/main/resources/archetype-resources/src/main/resources/default-regex-filters.txt @@ -14,11 +14,23 @@ # of services running on the crawling machine (e.g., Elasticsearch, # Storm) # -# - exclude localhost and loop-back addresses +# NOTE: these rules judge the bytes of the URL only. A host name which +# resolves into one of the ranges below is not caught here; the IP +# address filter (http.filter.ipaddress.exclude) sees the address the +# fetcher actually connects to and is the authoritative check. +# +# - exclude localhost and loop-back addresses, including the abbreviated +# and integer IPv4 forms the resolver maps to 127.0.0.1 # http://localhost:8080 # http://127.0.0.1/ .. http://127.255.255.255/ +# http://127.1/ +# http://2130706433/ # http://[::1]/ --^https?://(?:localhost|127(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){3}|\[::1\])(?::\d+)?(?:/|$) +-^https?://(?:localhost|127(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){0,3}|\[::1\])(?::\d+)?(?:/|$) +# the integer form of an IPv4 address resolves to the address it encodes +# (e.g. http://2130706433/ is 127.0.0.1); such URLs are blocked outright, +# a crawl that needs one can rewrite it to its dotted form +-^https?://\d{1,10}(?::\d+)?(?:/|$) # # - exclude private IP address spaces # 10.0.0.0/8 @@ -27,6 +39,14 @@ -^https?://(?:192\.168(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) # 172.16.0.0/12 -^https?://(?:172\.(?:1[6789]|2[0-9]|3[01])(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 169.254.0.0/16 (link-local, e.g. cloud instance metadata services) +-^https?://(?:169\.254(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 100.64.0.0/10 (carrier-grade NAT) +-^https?://(?:100\.(?:6[0-4]|[7-9][0-9]|1[01][0-9]|12[0-7])(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){2})(?::\d+)?(?:/|$) +# 0.0.0.0/8 ("this network") +-^https?://(?:0(?:\.(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?))){1,3})(?::\d+)?(?:/|$) +# IPv6 unique-local (fc00::/7) and link-local (fe80::/10) addresses +-^https?://\[(?:f[cd][0-9a-f]{2}|fe[89ab][0-9a-f]):[0-9a-f:]+\](?::\d+)?(?:/|$) # accept anything else +.