| # Licensed to the Apache Software Foundation (ASF) under one or more |
| # contributor license agreements. See the NOTICE file distributed with |
| # this work for additional information regarding copyright ownership. |
| # The ASF licenses this file to You under the Apache License, Version 2.0 |
| # (the "License"); you may not use this file except in compliance with |
| # the License. You may obtain a copy of the License at |
| # |
| # http://www.apache.org/licenses/LICENSE-2.0 |
| # |
| # Unless required by applicable law or agreed to in writing, software |
| # distributed under the License is distributed on an "AS IS" BASIS, |
| # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. |
| # See the License for the specific language governing permissions and |
| # limitations under the License. |
| |
| # Default configuration for Apache StormCrawler |
| # This is used to make the default values explicit and list the most common configurations. |
| # Do not modify this file but instead provide a custom one with the parameter -conf |
| # when launching your extension of ConfigurableTopology. |
| |
| config: |
| fetcher.server.delay: 1.0 |
| # min. delay for multi-threaded queues |
| fetcher.server.min.delay: 0.0 |
| fetcher.queue.mode: "byHost" |
| fetcher.threads.per.queue: 1 |
| fetcher.threads.number: 10 |
| fetcher.threads.start.delay: 10 |
| fetcher.max.urls.in.queues: -1 |
| fetcher.max.queue.size: -1 |
| fetcher.timeout.queue: -1 |
| # hard timeout in seconds for a single protocol fetch at the bolt level; |
| # -1 disables (relies on protocol-level socket timeouts only) |
| fetcher.thread.timeout: -1 |
| # max. crawl-delay accepted in robots.txt (in seconds) |
| fetcher.max.crawl.delay: 30 |
| # behavior of fetcher when the crawl-delay in the robots.txt |
| # is larger than fetcher.max.crawl.delay: |
| # (if false) |
| # skip URLs from this queue to avoid that any overlong |
| # crawl-delay throttles the crawler |
| # (if true) |
| # cap the local delay at fetcher.max.crawl.delay and emit the |
| # requested value in whole seconds as robots.crawl.delay; |
| # URLFrontier's QueueRegulatorBolt can forward a capped value when |
| # its robots crawl-delay integration is explicitly enabled. Without |
| # that consumer the fetcher remains more aggressive than requested. |
| fetcher.max.crawl.delay.force: false |
| # behavior of fetcher when the crawl-delay in the robots.txt |
| # is smaller (ev. less than one second) than the default delay: |
| # (if true) |
| # use the larger default delay (fetcher.server.delay) |
| # and ignore the shorter crawl-delay in the robots.txt |
| # (if false) |
| # use the delay specified in the robots.txt |
| fetcher.server.delay.force: false |
| |
| # time bucket to use for the metrics sent by the Fetcher |
| fetcher.metrics.time.bucket.secs: 10 |
| |
| # metrics version: "v1" (legacy), "v2" (Storm V2 native), or "both" |
| stormcrawler.metrics.version: "v1" |
| |
| # SimpleFetcherBolt: if the delay required by the politeness |
| # is above this value, the tuple is sent back to the Storm queue |
| # for the bolt on the _throttle_ stream (in msec) |
| fetcher.max.throttle.sleep: -1 |
| |
| # alternative values are "byIP" and "byDomain" |
| partition.url.mode: "byHost" |
| |
| urlbuffer.class: "org.apache.stormcrawler.persistence.urlbuffer.SimpleURLBuffer" |
| |
| # Lists the metadata to transfer to outlinks |
| # Used by Fetcher and SiteMapParser for redirections, |
| # discovered links, passing cookies to child pages |
| # (protocol.set-cookie and protocol.set-cookie-origin, |
| # which have to be transferred together), etc. |
| # These are also persisted for the parent document (see below). |
| # Allows wildcards, eg. "follow.*" transfers all metadata starting with "follow.". |
| # metadata.transfer: |
| # - customMetadataName |
| |
| # Lists the metadata to persist to storage |
| # These are not transferred to the outlinks. Also allows wildcards, eg. "follow.*". |
| metadata.persist: |
| - _redirTo |
| - error.cause |
| - error.source |
| - isSitemap |
| - isFeed |
| |
| metadata.track.path: true |
| metadata.track.depth: true |
| |
| # Agent name info - given here as an example. Do not be an anonymous coward, use your real information! |
| # The full user agent value sent as part of the HTTP requests |
| # is built from the elements below. Only the agent.name is mandatory, |
| # it is also used to parse the robots.txt directives. |
| |
| # The agent name must be compliant with RFC 9309 (section 2.2.1) |
| # i.e. it MUST contain only uppercase and lowercase letters ("a-z" and "A-Z), underscores ("_"), and hyphens ("-") |
| # http.agent.name: "AnonymousCoward" |
| # version of your crawler |
| # http.agent.version: "1.0" |
| # description of what it does |
| # http.agent.description: "built with StormCrawler" |
| # URL webmasters can go to to learn about it |
| # http.agent.url: "http://someorganization.com/" |
| # Finally, an email so that they can get in touch with you |
| # http.agent.email: "someone@someorganization.com" |
| |
| # user-agent name(s), used to select rules from the |
| # robots.txt file by matching the names against the user-agent |
| # lines in the robots.txt file. Optional, if empty, the value |
| # of http.agent.name is used. Otherwise, it must be listed first. |
| # the tokens must be compliant with RFC 9309 (section 2.2.1). |
| # http.robots.agents: agents as a comma separated string but can also take a list |
| |
| # (advanced) Specify the user agent to send to the HTTP requests |
| # note that this is not used for parsing the robots.txt and |
| # therefore you need to have set _http.agent.name_. |
| # http.agent: "Verbatim user agent" |
| |
| http.accept.language: "en-us,en-gb,en;q=0.7,*;q=0.3" |
| http.accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8" |
| http.content.limit: -1 |
| http.store.headers: false |
| http.timeout: 10000 |
| |
| # store partial fetches as trimmed content (some content has been fetched, |
| # but reading more data from socket failed, eg. because of a network timeout) |
| http.content.partial.as.trimmed: false |
| |
| # for crawling through a proxy: |
| # 1-line config |
| # http.proxy: "http://localhost:8000" |
| # http.proxy.host: |
| # http.proxy.port: |
| # okhttp only, defaults to "HTTP" |
| # http.proxy.type: "SOCKS" |
| # for crawling through a proxy with Basic authentication: |
| # http.proxy.user: |
| # http.proxy.pass: |
| |
| # Retry on connection failure: |
| http.retry.on.connection.failure: true |
| |
| # Follow redirect HTTP responses: |
| http.allow.redirects: false |
| |
| # Accept any TLS certificate, including self-signed, expired or otherwise |
| # invalid ones (okhttp protocol only)? The certificate chains are accepted |
| # without validation, i.e. the servers are not authenticated: anyone able to |
| # answer for the host name receives everything sent to them. Needed for |
| # crawling hosts with unvalidatable certificates, e.g. self-signed ones, but |
| # also disables protection against man-in-the-middle attacks. Credentials |
| # (basic auth, credential headers, cookies) are withheld from cleartext |
| # http:// requests and from https:// requests on such connections unless |
| # http.credentials.allow.insecure is set to true. |
| http.trust.everything: false |
| |
| # Check that the certificate presented by the server matches the host name |
| # contacted (okhttp protocol only)? Disabling this is independent of |
| # http.trust.everything: a valid certificate for a different name is then |
| # accepted. Hostname verification does not need to be disabled for hosts |
| # with self-signed certificates when http.trust.everything is enabled. When |
| # disabled, credentials are withheld unless http.credentials.allow.insecure |
| # is set to true. |
| http.verify.hostnames: true |
| |
| # Send credentials (basic auth, credential headers, cookies) even on |
| # connections which do not authenticate the server (okhttp protocol only)? |
| # Without it, credentials are withheld from cleartext http:// requests and |
| # from https:// requests whose certificate was not validated |
| # (http.trust.everything) or not checked against the host name |
| # (http.verify.hostnames: false), so that they are not disclosed to servers |
| # which were never authenticated. |
| http.credentials.allow.insecure: false |
| |
| # Additional header names (case-insensitive) which carry credentials and are |
| # therefore withheld from requests to servers which were not authenticated |
| # (okhttp protocol only, see http.trust.everything and |
| # http.credentials.allow.insecure). Applies to the headers configured with |
| # http.custom.headers and to headers set by request; cookies are withheld |
| # separately. "authorization", "proxy-authorization", "cookie" and |
| # "x-api-key" are always treated as credentials and cannot be removed here. |
| # http.credentials.headers: |
| # - "x-auth-token" |
| |
| # Basic authentication for the sites which require it. The Authorization |
| # header built from http.basicauth.user and http.basicauth.password is only |
| # sent to the hosts listed in http.basicauth.hosts (a list or a |
| # comma-separated string): host names or IP addresses without scheme, port, |
| # path or wildcard, matched exactly and case-insensitively; other entries are |
| # ignored with a warning. Subdomains are not included, list each host |
| # separately. The host of every request is checked, robots.txt fetches and |
| # redirect targets included; a redirect followed with http.allow.redirects |
| # never carries the header to another scheme, host or port. The header is |
| # also withheld from servers which were not authenticated, see |
| # http.credentials.allow.insecure. |
| # Breaking change in 4.0.0: the credentials used to be sent to every host |
| # crawled. If http.basicauth.hosts is empty they are not sent at all and a |
| # warning is logged at startup. |
| # Credential headers configured with http.custom.headers are still sent to |
| # every host; set them per site with protocol.set-header in the metadata of |
| # the URLs instead. |
| # http.basicauth.user: |
| # http.basicauth.password: |
| # http.basicauth.hosts: |
| # - "intranet.example.com" |
| |
| # Maximum number of redirect hops followed when http.allow.redirects is |
| # enabled (okhttp protocol only). A chain which does not end within this |
| # many hops returns its last redirect response, which the caller handles |
| # like it does when redirects are not followed at all. |
| http.allow.redirects.max: 5 |
| |
| # IP address filtering (okhttp protocol only). Optionally limit or block |
| # connections to IP address ranges once the host name has been resolved. This |
| # prevents information leakage to a public index when a DNS entry points to a |
| # private or loopback address. Rules can be given as a comma-separated string |
| # or as a list and may be: |
| # - a single IP address, e.g. "127.0.0.1" or "::1" |
| # - a CIDR block, e.g. "192.168.0.0/16" or "fd00::/8" |
| # - "localhost" / "loopback" (matches InetAddress.isLoopbackAddress()) |
| # - "sitelocal" (matches InetAddress.isSiteLocalAddress()) |
| # - "linklocal" (matches InetAddress.isLinkLocalAddress()) |
| # - "anylocal" (matches InetAddress.isAnyLocalAddress()) |
| # - "multicast" (matches InetAddress.isMulticastAddress()) |
| # A value which is neither of these is a configuration error and stops the |
| # topology from starting. |
| # Only addresses matching an include rule are fetched (empty means all are |
| # allowed), addresses matching an exclude rule are always blocked. |
| # The exclude list is enabled by default: a fetched page decides which hosts |
| # the fetcher connects to, and loopback, private and link-local ranges host |
| # unauthenticated services (e.g. cloud instance metadata) which a public |
| # index must not leak into. To crawl an intranet, override the exclude list |
| # explicitly (exclude rules win over include rules, so setting only |
| # http.filter.ipaddress.include does not undo this default), e.g. |
| # http.filter.ipaddress.exclude: "" |
| # Fetches through a proxy are not filtered: the proxy resolves and connects |
| # to the target, so only its address is known here and a proxy on a private |
| # address is fine. The proxy's own egress rules decide what it may reach. |
| http.filter.ipaddress.exclude: "localhost,sitelocal,linklocal,anylocal,multicast,100.64.0.0/10,0.0.0.0/8,fc00::/7,::/128" |
| |
| # Allow all if robots.txt cannot be parsed due to code 403 (Forbidden): |
| http.robots.403.allow: true |
| |
| # Allow all if robots.txt cannot be parsed due to a server error (5xx): |
| http.robots.5xx.allow: false |
| |
| # Follow a robots.txt redirect whose target is on a different scheme, host or |
| # port than the URL it was reached from? Redirects to schemes other than http |
| # and https are never followed. A redirect which only replaces http with https |
| # while staying on the same host and port is always followed, whatever this is |
| # set to. When this is enabled, the robots.txt is re-fetched from the host |
| # named in the Location header, for up to 5 hops, and each of these requests |
| # carries the headers configured through http.custom.headers, and those of |
| # http.basicauth.* if the host is listed in http.basicauth.hosts, as any other |
| # request does; http.filter.ipaddress.exclude |
| # can be used to restrict the addresses those fetches may reach. |
| http.robots.redirect.crossorigin.allow: false |
| |
| # Allow all if a robots.txt redirect was not followed because of the setting |
| # above? If false, nothing is crawled on that host until the redirect is |
| # followed or this is set to true. |
| http.robots.redirect.refused.allow: false |
| |
| # ignore directives from robots.txt files? |
| http.robots.file.skip: false |
| |
| # ignore robots directives from the http headers? |
| http.robots.headers.skip: false |
| |
| # ignore robots directives from the html meta? |
| http.robots.meta.skip: false |
| |
| # should the URLs be removed when a page is marked as noFollow |
| robots.noFollow.strict: true |
| |
| # http.content.limit when fetching the robots.txt |
| # (the robots.txt RFC draft requires to fetch and parse at least 500 kiB, |
| # see https://datatracker.ietf.org/doc/html/draft-rep-wg-topic-00#section-2.5) |
| # A value of -1 means "same as http.content.limit": the robots.txt fetch |
| # then uses whatever limit is configured for pages, including "no limit" |
| # when http.content.limit is -1 too. |
| # http.robots.content.limit: 524288 # 512 kiB |
| http.robots.content.limit: 524288 |
| |
| # Implementation of RobotRulesParser used by the HTTP protocol implementations |
| # to fetch and parse robots.txt. Override to plug in custom robots.txt |
| # directive handling (e.g. parsing extensions not supported by crawler-commons). |
| # The class must extend org.apache.stormcrawler.protocol.RobotRulesParser and |
| # have a public no-arg constructor. |
| http.robots.parser.class: "org.apache.stormcrawler.protocol.HttpRobotRulesParser" |
| |
| # Guava caches used for the robots.txt directives |
| robots.cache.spec: "maximumSize=10000,expireAfterWrite=6h" |
| robots.error.cache.spec: "maximumSize=10000,expireAfterWrite=1h" |
| |
| # The file scheme must be enabled deliberately: a fetched page can put a |
| # file:// URL into the frontier, and FileProtocol reads whatever path the |
| # worker user can read unless file.protocol.root confines it. When enabled, |
| # set file.protocol.root to the directory reads are confined to; without it |
| # the file scheme serves nothing. |
| protocols: "http,https" |
| http.protocol.implementation: "org.apache.stormcrawler.protocol.okhttp.HttpProtocol" |
| https.protocol.implementation: "org.apache.stormcrawler.protocol.okhttp.HttpProtocol" |
| file.protocol.implementation: "org.apache.stormcrawler.protocol.file.FileProtocol" |
| # file.protocol.root: "/data/corpus" |
| |
| # number of instances for each protocol implementation |
| protocol.instances.num: 1 |
| |
| # the http/https protocol versions to use, in order of preference |
| # Details of the protocol negotiation between the client and |
| # the crawled server depend on the chosen protocol implementation. |
| # If no protocol versions are listed the protocol implementation |
| # will use its defaults. |
| http.protocol.versions: |
| # HTTP/2 over TLS (protocol negotiation via ALPN) |
| #- "h2" |
| # HTTP/1.1 |
| #- "http/1.1" |
| # HTTP/1.0 |
| #- "http/1.0" |
| # HTTP/2 over TCP |
| ##- "h2c" |
| |
| # connection pool configuration of OkHttp protocol |
| okhttp.protocol.connection.pool: |
| # maximum number of idle connections (in addition to active connections) |
| max.idle.connections: 5 |
| # maximum keep-alive time of the connections in seconds |
| connection.keep.alive: 300 |
| # See also |
| # https://square.github.io/okhttp/3.x/okhttp/okhttp3/ConnectionPool.html |
| # Note that OkHttp's connection pool (v4.9.1) is not optimized for fast |
| # look-up of connections, the pool size (idle and active connections) |
| # should not exceed 1000. To allow for efficient pooling in large and |
| # diverse crawls, it's recommended to increase also the number of protocol |
| # instances, see `protocol.instances.num`. |
| |
| # key values obtained by the protocol can be prefixed |
| # to avoid accidental overwrites. Note that persisted |
| # or transferred protocol metadata must also be prefixed. |
| protocol.md.prefix: "protocol." |
| |
| # no url or parsefilters by default |
| # parsefilters.config.file: "parsefilters.json" |
| # urlfilters.config.file: "urlfilters.json" |
| |
| # JSoupParserBolt |
| jsoup.treat.non.html.as.error: true |
| parser.emitOutlinks: true |
| parser.emitOutlinks.max.per.page: -1 |
| track.anchors: true |
| detect.mimetype: true |
| detect.charset.maxlength: 10000 |
| |
| #textextractor.class: "org.apache.stormcrawler.parse.JSoupTextExtractor" |
| textextractor.skip.after: -1 |
| |
| # filters URLs in sitemaps based on their modified Date (if any) |
| sitemap.filter.hours.since.modified: -1 |
| |
| # whether a document without the isSitemap key is classified as a sitemap |
| # by searching the first bytes of its content for the sitemaps.org |
| # namespace. Off by default: any page carrying the namespace string early |
| # enough would be reclassified as a sitemap and never reach the parser |
| # bolt. When enabled, a content type which rules a sitemap out (a page |
| # served as HTML) stops the sniffing. |
| sitemap.sniffContent: false |
| |
| # whether the sitemap parser applies strict URL checking: a sitemap then |
| # only yields URLs below its own host and path (strict URL checking of |
| # crawler-commons, not its namespace check), so a sitemap cannot enrol URLs |
| # on hosts it has nothing to do with. Applied to <urlset> entries by the |
| # parser and to the <loc> entries of a sitemap index by this bolt. Off by |
| # default: a sitemap living at example.com while listing URLs under |
| # www.example.com violates the sitemap spec but is common, and strict |
| # checking would silently shrink such a crawl. Recommended for open crawls. |
| sitemap.strict: false |
| |
| # staggered scheduling of sitemaps |
| sitemap.schedule.delay: -1 |
| |
| # whether to add any sitemaps found in the robots.txt to the status stream |
| # used by fetcher bolts |
| sitemap.discovery: false |
| |
| # determines what sitemap extensions to parse from the sitemap and add |
| # to an outlinks metadata object |
| sitemap.extensions: |
| # Illustrates enabling sitemap extension parsing |
| # there are 5 supported types "IMAGE", "LINKS", "MOBILE", "NEWS", and "VIDEO" |
| # sitemap.extensions: |
| # - IMAGE |
| # - LINKS |
| # - MOBILE |
| # - NEWS |
| # - VIDEO |
| |
| # Default implementation of Scheduler |
| scheduler.class: "org.apache.stormcrawler.persistence.DefaultScheduler" |
| |
| # revisit a page daily (value in minutes) |
| # set it to -1 to never refetch a page |
| fetchInterval.default: 1440 |
| |
| # revisit a page with a fetch error after 2 hours (value in minutes) |
| # set it to -1 to never refetch a page |
| fetchInterval.fetch.error: 120 |
| |
| # never revisit a page with an error (or set a value in minutes) |
| fetchInterval.error: -1 |
| |
| # custom fetch interval to be used when a document has the key/value in its metadata |
| # and has been fetched successfully (value in minutes) |
| # fetchInterval.FETCH_ERROR.isFeed=true |
| # fetchInterval.isFeed=true: 10 |
| |
| # max number of successive fetch errors before changing status to ERROR |
| max.fetch.errors: 3 |
| |
| # Guava cache use by AbstractStatusUpdaterBolt for DISCOVERED URLs |
| status.updater.use.cache: true |
| status.updater.cache.spec: "expireAfterAccess=1h,softValues" |
| |
| # Normalise the host component of DISCOVERED URLs before they are stored, |
| # so that aliases of one server (percent-escaping, case, trailing dot) |
| # become a single record, a single politeness queue and a single robots.txt |
| # fetch. Off by default: document ids are hashed from the URL, so turning |
| # this on makes servers whose aliases were already stored in raw form be |
| # re-discovered under the normalised form as a second, bounded record. |
| status.updater.normalise.hosts: false |
| |
| # Can also take "MINUTE" or "HOUR" |
| status.updater.unit.round.date: "SECOND" |
| |
| # configuration for the classes extending AbstractIndexerBolt |
| # indexer.md.filter: "someKey=aValue" |
| indexer.md.docid: "" |
| indexer.ignore.empty.fields: false |
| indexer.url.fieldname: "url" |
| indexer.text.fieldname: "content" |
| indexer.text.maxlength: -1 |
| indexer.canonical.name: "canonical" |
| # How to convert metadata key values into fields for indexing |
| # |
| # if no alias is specified with =alias, the key value is used |
| # for instance below, _domain_ and _format_ will be used |
| # as field names, whereas _title_ will be used for _parse.title_. |
| # You can specify the index of the value to store from the values array |
| # by using the _key[index]_ format, e.g. _parse.title[0]_ would try to |
| # get the first value for the metadata _parse.title_ (which is the default anyway). |
| # Finally, you can use a glob (*) to match all the keys, e.g. _parse.*_ would |
| # index all the keys with _parse_ as a prefix. Note that in that case, you can't |
| # specify an alias with =, nor can you specify an index. |
| indexer.md.mapping: |
| - parse.title=title |
| - parse.keywords=keywords |
| - parse.description=description |
| |