blob: da5e32265982dc2cea43aa0ab296d4c6804981a1 [file]
# Licensed to the Apache Software Foundation (ASF) under one or more
# contributor license agreements. See the NOTICE file distributed with
# this work for additional information regarding copyright ownership.
# The ASF licenses this file to You under the Apache License, Version 2.0
# (the "License"); you may not use this file except in compliance with
# the License. You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# Default configuration for Apache StormCrawler
# This is used to make the default values explicit and list the most common configurations.
# Do not modify this file but instead provide a custom one with the parameter -conf
# when launching your extension of ConfigurableTopology.
config:
fetcher.server.delay: 1.0
# min. delay for multi-threaded queues
fetcher.server.min.delay: 0.0
fetcher.queue.mode: "byHost"
fetcher.threads.per.queue: 1
fetcher.threads.number: 10
fetcher.threads.start.delay: 10
fetcher.max.urls.in.queues: -1
fetcher.max.queue.size: -1
fetcher.timeout.queue: -1
# hard timeout in seconds for a single protocol fetch at the bolt level;
# -1 disables (relies on protocol-level socket timeouts only)
fetcher.thread.timeout: -1
# max. crawl-delay accepted in robots.txt (in seconds)
fetcher.max.crawl.delay: 30
# behavior of fetcher when the crawl-delay in the robots.txt
# is larger than fetcher.max.crawl.delay:
# (if false)
# skip URLs from this queue to avoid that any overlong
# crawl-delay throttles the crawler
# (if true)
# cap the local delay at fetcher.max.crawl.delay and emit the
# requested value in whole seconds as robots.crawl.delay;
# URLFrontier's QueueRegulatorBolt can forward a capped value when
# its robots crawl-delay integration is explicitly enabled. Without
# that consumer the fetcher remains more aggressive than requested.
fetcher.max.crawl.delay.force: false
# behavior of fetcher when the crawl-delay in the robots.txt
# is smaller (ev. less than one second) than the default delay:
# (if true)
# use the larger default delay (fetcher.server.delay)
# and ignore the shorter crawl-delay in the robots.txt
# (if false)
# use the delay specified in the robots.txt
fetcher.server.delay.force: false
# time bucket to use for the metrics sent by the Fetcher
fetcher.metrics.time.bucket.secs: 10
# metrics version: "v1" (legacy), "v2" (Storm V2 native), or "both"
stormcrawler.metrics.version: "v1"
# SimpleFetcherBolt: if the delay required by the politeness
# is above this value, the tuple is sent back to the Storm queue
# for the bolt on the _throttle_ stream (in msec)
fetcher.max.throttle.sleep: -1
# alternative values are "byIP" and "byDomain"
partition.url.mode: "byHost"
urlbuffer.class: "org.apache.stormcrawler.persistence.urlbuffer.SimpleURLBuffer"
# Lists the metadata to transfer to outlinks
# Used by Fetcher and SiteMapParser for redirections,
# discovered links, passing cookies to child pages
# (protocol.set-cookie and protocol.set-cookie-origin,
# which have to be transferred together), etc.
# These are also persisted for the parent document (see below).
# Allows wildcards, eg. "follow.*" transfers all metadata starting with "follow.".
# metadata.transfer:
# - customMetadataName
# Lists the metadata to persist to storage
# These are not transferred to the outlinks. Also allows wildcards, eg. "follow.*".
metadata.persist:
- _redirTo
- error.cause
- error.source
- isSitemap
- isFeed
metadata.track.path: true
metadata.track.depth: true
# Agent name info - given here as an example. Do not be an anonymous coward, use your real information!
# The full user agent value sent as part of the HTTP requests
# is built from the elements below. Only the agent.name is mandatory,
# it is also used to parse the robots.txt directives.
# The agent name must be compliant with RFC 9309 (section 2.2.1)
# i.e. it MUST contain only uppercase and lowercase letters ("a-z" and "A-Z), underscores ("_"), and hyphens ("-")
# http.agent.name: "AnonymousCoward"
# version of your crawler
# http.agent.version: "1.0"
# description of what it does
# http.agent.description: "built with StormCrawler"
# URL webmasters can go to to learn about it
# http.agent.url: "http://someorganization.com/"
# Finally, an email so that they can get in touch with you
# http.agent.email: "someone@someorganization.com"
# user-agent name(s), used to select rules from the
# robots.txt file by matching the names against the user-agent
# lines in the robots.txt file. Optional, if empty, the value
# of http.agent.name is used. Otherwise, it must be listed first.
# the tokens must be compliant with RFC 9309 (section 2.2.1).
# http.robots.agents: agents as a comma separated string but can also take a list
# (advanced) Specify the user agent to send to the HTTP requests
# note that this is not used for parsing the robots.txt and
# therefore you need to have set _http.agent.name_.
# http.agent: "Verbatim user agent"
http.accept.language: "en-us,en-gb,en;q=0.7,*;q=0.3"
http.accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
http.content.limit: -1
http.store.headers: false
http.timeout: 10000
# store partial fetches as trimmed content (some content has been fetched,
# but reading more data from socket failed, eg. because of a network timeout)
http.content.partial.as.trimmed: false
# for crawling through a proxy:
# 1-line config
# http.proxy: "http://localhost:8000"
# http.proxy.host:
# http.proxy.port:
# okhttp only, defaults to "HTTP"
# http.proxy.type: "SOCKS"
# for crawling through a proxy with Basic authentication:
# http.proxy.user:
# http.proxy.pass:
# Retry on connection failure:
http.retry.on.connection.failure: true
# Follow redirect HTTP responses:
http.allow.redirects: false
# Accept any TLS certificate, including self-signed, expired or otherwise
# invalid ones (okhttp protocol only)? The certificate chains are accepted
# without validation, i.e. the servers are not authenticated: anyone able to
# answer for the host name receives everything sent to them. Needed for
# crawling hosts with unvalidatable certificates, e.g. self-signed ones, but
# also disables protection against man-in-the-middle attacks. Credentials
# (basic auth, credential headers, cookies) are withheld from cleartext
# http:// requests and from https:// requests on such connections unless
# http.credentials.allow.insecure is set to true.
http.trust.everything: false
# Check that the certificate presented by the server matches the host name
# contacted (okhttp protocol only)? Disabling this is independent of
# http.trust.everything: a valid certificate for a different name is then
# accepted. Hostname verification does not need to be disabled for hosts
# with self-signed certificates when http.trust.everything is enabled. When
# disabled, credentials are withheld unless http.credentials.allow.insecure
# is set to true.
http.verify.hostnames: true
# Send credentials (basic auth, credential headers, cookies) even on
# connections which do not authenticate the server (okhttp protocol only)?
# Without it, credentials are withheld from cleartext http:// requests and
# from https:// requests whose certificate was not validated
# (http.trust.everything) or not checked against the host name
# (http.verify.hostnames: false), so that they are not disclosed to servers
# which were never authenticated.
http.credentials.allow.insecure: false
# Additional header names (case-insensitive) which carry credentials and are
# therefore withheld from requests to servers which were not authenticated
# (okhttp protocol only, see http.trust.everything and
# http.credentials.allow.insecure). Applies to the headers configured with
# http.custom.headers and to headers set by request; cookies are withheld
# separately. "authorization", "proxy-authorization", "cookie" and
# "x-api-key" are always treated as credentials and cannot be removed here.
# http.credentials.headers:
# - "x-auth-token"
# Basic authentication for the sites which require it. The Authorization
# header built from http.basicauth.user and http.basicauth.password is only
# sent to the hosts listed in http.basicauth.hosts (a list or a
# comma-separated string): host names or IP addresses without scheme, port,
# path or wildcard, matched exactly and case-insensitively; other entries are
# ignored with a warning. Subdomains are not included, list each host
# separately. The host of every request is checked, robots.txt fetches and
# redirect targets included; a redirect followed with http.allow.redirects
# never carries the header to another scheme, host or port. The header is
# also withheld from servers which were not authenticated, see
# http.credentials.allow.insecure.
# Breaking change in 4.0.0: the credentials used to be sent to every host
# crawled. If http.basicauth.hosts is empty they are not sent at all and a
# warning is logged at startup.
# Credential headers configured with http.custom.headers are still sent to
# every host; set them per site with protocol.set-header in the metadata of
# the URLs instead.
# http.basicauth.user:
# http.basicauth.password:
# http.basicauth.hosts:
# - "intranet.example.com"
# Maximum number of redirect hops followed when http.allow.redirects is
# enabled (okhttp protocol only). A chain which does not end within this
# many hops returns its last redirect response, which the caller handles
# like it does when redirects are not followed at all.
http.allow.redirects.max: 5
# IP address filtering (okhttp protocol only). Optionally limit or block
# connections to IP address ranges once the host name has been resolved. This
# prevents information leakage to a public index when a DNS entry points to a
# private or loopback address. Rules can be given as a comma-separated string
# or as a list and may be:
# - a single IP address, e.g. "127.0.0.1" or "::1"
# - a CIDR block, e.g. "192.168.0.0/16" or "fd00::/8"
# - "localhost" / "loopback" (matches InetAddress.isLoopbackAddress())
# - "sitelocal" (matches InetAddress.isSiteLocalAddress())
# - "linklocal" (matches InetAddress.isLinkLocalAddress())
# - "anylocal" (matches InetAddress.isAnyLocalAddress())
# - "multicast" (matches InetAddress.isMulticastAddress())
# A value which is neither of these is a configuration error and stops the
# topology from starting.
# Only addresses matching an include rule are fetched (empty means all are
# allowed), addresses matching an exclude rule are always blocked.
# The exclude list is enabled by default: a fetched page decides which hosts
# the fetcher connects to, and loopback, private and link-local ranges host
# unauthenticated services (e.g. cloud instance metadata) which a public
# index must not leak into. To crawl an intranet, override the exclude list
# explicitly (exclude rules win over include rules, so setting only
# http.filter.ipaddress.include does not undo this default), e.g.
# http.filter.ipaddress.exclude: ""
# Fetches through a proxy are not filtered: the proxy resolves and connects
# to the target, so only its address is known here and a proxy on a private
# address is fine. The proxy's own egress rules decide what it may reach.
http.filter.ipaddress.exclude: "localhost,sitelocal,linklocal,anylocal,multicast,100.64.0.0/10,0.0.0.0/8,fc00::/7,::/128"
# Allow all if robots.txt cannot be parsed due to code 403 (Forbidden):
http.robots.403.allow: true
# Allow all if robots.txt cannot be parsed due to a server error (5xx):
http.robots.5xx.allow: false
# Follow a robots.txt redirect whose target is on a different scheme, host or
# port than the URL it was reached from? Redirects to schemes other than http
# and https are never followed. A redirect which only replaces http with https
# while staying on the same host and port is always followed, whatever this is
# set to. When this is enabled, the robots.txt is re-fetched from the host
# named in the Location header, for up to 5 hops, and each of these requests
# carries the headers configured through http.custom.headers, and those of
# http.basicauth.* if the host is listed in http.basicauth.hosts, as any other
# request does; http.filter.ipaddress.exclude
# can be used to restrict the addresses those fetches may reach.
http.robots.redirect.crossorigin.allow: false
# Allow all if a robots.txt redirect was not followed because of the setting
# above? If false, nothing is crawled on that host until the redirect is
# followed or this is set to true.
http.robots.redirect.refused.allow: false
# ignore directives from robots.txt files?
http.robots.file.skip: false
# ignore robots directives from the http headers?
http.robots.headers.skip: false
# ignore robots directives from the html meta?
http.robots.meta.skip: false
# should the URLs be removed when a page is marked as noFollow
robots.noFollow.strict: true
# http.content.limit when fetching the robots.txt
# (the robots.txt RFC draft requires to fetch and parse at least 500 kiB,
# see https://datatracker.ietf.org/doc/html/draft-rep-wg-topic-00#section-2.5)
# A value of -1 means "same as http.content.limit": the robots.txt fetch
# then uses whatever limit is configured for pages, including "no limit"
# when http.content.limit is -1 too.
# http.robots.content.limit: 524288 # 512 kiB
http.robots.content.limit: 524288
# Implementation of RobotRulesParser used by the HTTP protocol implementations
# to fetch and parse robots.txt. Override to plug in custom robots.txt
# directive handling (e.g. parsing extensions not supported by crawler-commons).
# The class must extend org.apache.stormcrawler.protocol.RobotRulesParser and
# have a public no-arg constructor.
http.robots.parser.class: "org.apache.stormcrawler.protocol.HttpRobotRulesParser"
# Guava caches used for the robots.txt directives
robots.cache.spec: "maximumSize=10000,expireAfterWrite=6h"
robots.error.cache.spec: "maximumSize=10000,expireAfterWrite=1h"
# The file scheme must be enabled deliberately: a fetched page can put a
# file:// URL into the frontier, and FileProtocol reads whatever path the
# worker user can read unless file.protocol.root confines it. When enabled,
# set file.protocol.root to the directory reads are confined to; without it
# the file scheme serves nothing.
protocols: "http,https"
http.protocol.implementation: "org.apache.stormcrawler.protocol.okhttp.HttpProtocol"
https.protocol.implementation: "org.apache.stormcrawler.protocol.okhttp.HttpProtocol"
file.protocol.implementation: "org.apache.stormcrawler.protocol.file.FileProtocol"
# file.protocol.root: "/data/corpus"
# number of instances for each protocol implementation
protocol.instances.num: 1
# the http/https protocol versions to use, in order of preference
# Details of the protocol negotiation between the client and
# the crawled server depend on the chosen protocol implementation.
# If no protocol versions are listed the protocol implementation
# will use its defaults.
http.protocol.versions:
# HTTP/2 over TLS (protocol negotiation via ALPN)
#- "h2"
# HTTP/1.1
#- "http/1.1"
# HTTP/1.0
#- "http/1.0"
# HTTP/2 over TCP
##- "h2c"
# connection pool configuration of OkHttp protocol
okhttp.protocol.connection.pool:
# maximum number of idle connections (in addition to active connections)
max.idle.connections: 5
# maximum keep-alive time of the connections in seconds
connection.keep.alive: 300
# See also
# https://square.github.io/okhttp/3.x/okhttp/okhttp3/ConnectionPool.html
# Note that OkHttp's connection pool (v4.9.1) is not optimized for fast
# look-up of connections, the pool size (idle and active connections)
# should not exceed 1000. To allow for efficient pooling in large and
# diverse crawls, it's recommended to increase also the number of protocol
# instances, see `protocol.instances.num`.
# key values obtained by the protocol can be prefixed
# to avoid accidental overwrites. Note that persisted
# or transferred protocol metadata must also be prefixed.
protocol.md.prefix: "protocol."
# no url or parsefilters by default
# parsefilters.config.file: "parsefilters.json"
# urlfilters.config.file: "urlfilters.json"
# JSoupParserBolt
jsoup.treat.non.html.as.error: true
parser.emitOutlinks: true
parser.emitOutlinks.max.per.page: -1
track.anchors: true
detect.mimetype: true
detect.charset.maxlength: 10000
#textextractor.class: "org.apache.stormcrawler.parse.JSoupTextExtractor"
textextractor.skip.after: -1
# filters URLs in sitemaps based on their modified Date (if any)
sitemap.filter.hours.since.modified: -1
# whether a document without the isSitemap key is classified as a sitemap
# by searching the first bytes of its content for the sitemaps.org
# namespace. Off by default: any page carrying the namespace string early
# enough would be reclassified as a sitemap and never reach the parser
# bolt. When enabled, a content type which rules a sitemap out (a page
# served as HTML) stops the sniffing.
sitemap.sniffContent: false
# whether the sitemap parser applies strict URL checking: a sitemap then
# only yields URLs below its own host and path (strict URL checking of
# crawler-commons, not its namespace check), so a sitemap cannot enrol URLs
# on hosts it has nothing to do with. Applied to <urlset> entries by the
# parser and to the <loc> entries of a sitemap index by this bolt. Off by
# default: a sitemap living at example.com while listing URLs under
# www.example.com violates the sitemap spec but is common, and strict
# checking would silently shrink such a crawl. Recommended for open crawls.
sitemap.strict: false
# staggered scheduling of sitemaps
sitemap.schedule.delay: -1
# whether to add any sitemaps found in the robots.txt to the status stream
# used by fetcher bolts
sitemap.discovery: false
# determines what sitemap extensions to parse from the sitemap and add
# to an outlinks metadata object
sitemap.extensions:
# Illustrates enabling sitemap extension parsing
# there are 5 supported types "IMAGE", "LINKS", "MOBILE", "NEWS", and "VIDEO"
# sitemap.extensions:
# - IMAGE
# - LINKS
# - MOBILE
# - NEWS
# - VIDEO
# Default implementation of Scheduler
scheduler.class: "org.apache.stormcrawler.persistence.DefaultScheduler"
# revisit a page daily (value in minutes)
# set it to -1 to never refetch a page
fetchInterval.default: 1440
# revisit a page with a fetch error after 2 hours (value in minutes)
# set it to -1 to never refetch a page
fetchInterval.fetch.error: 120
# never revisit a page with an error (or set a value in minutes)
fetchInterval.error: -1
# custom fetch interval to be used when a document has the key/value in its metadata
# and has been fetched successfully (value in minutes)
# fetchInterval.FETCH_ERROR.isFeed=true
# fetchInterval.isFeed=true: 10
# max number of successive fetch errors before changing status to ERROR
max.fetch.errors: 3
# Guava cache use by AbstractStatusUpdaterBolt for DISCOVERED URLs
status.updater.use.cache: true
status.updater.cache.spec: "expireAfterAccess=1h,softValues"
# Normalise the host component of DISCOVERED URLs before they are stored,
# so that aliases of one server (percent-escaping, case, trailing dot)
# become a single record, a single politeness queue and a single robots.txt
# fetch. Off by default: document ids are hashed from the URL, so turning
# this on makes servers whose aliases were already stored in raw form be
# re-discovered under the normalised form as a second, bounded record.
status.updater.normalise.hosts: false
# Can also take "MINUTE" or "HOUR"
status.updater.unit.round.date: "SECOND"
# configuration for the classes extending AbstractIndexerBolt
# indexer.md.filter: "someKey=aValue"
indexer.md.docid: ""
indexer.ignore.empty.fields: false
indexer.url.fieldname: "url"
indexer.text.fieldname: "content"
indexer.text.maxlength: -1
indexer.canonical.name: "canonical"
# How to convert metadata key values into fields for indexing
#
# if no alias is specified with =alias, the key value is used
# for instance below, _domain_ and _format_ will be used
# as field names, whereas _title_ will be used for _parse.title_.
# You can specify the index of the value to store from the values array
# by using the _key[index]_ format, e.g. _parse.title[0]_ would try to
# get the first value for the metadata _parse.title_ (which is the default anyway).
# Finally, you can use a glob (*) to match all the keys, e.g. _parse.*_ would
# index all the keys with _parse_ as a prefix. Note that in that case, you can't
# specify an alias with =, nor can you specify an index.
indexer.md.mapping:
- parse.title=title
- parse.keywords=keywords
- parse.description=description