| # Licensed to the Apache Software Foundation (ASF) under one |
| # or more contributor license agreements. See the NOTICE file |
| # distributed with this work for additional information |
| # regarding copyright ownership. The ASF licenses this file |
| # to you under the Apache License, Version 2.0 (the |
| # "License"); you may not use this file except in compliance |
| # with the License. You may obtain a copy of the License at |
| # |
| # http://www.apache.org/licenses/LICENSE-2.0 |
| # |
| # Unless required by applicable law or agreed to in writing, software |
| # distributed under the License is distributed on an "AS IS" BASIS, |
| # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. |
| # See the License for the specific language governing permissions and |
| # limitations under the License. |
| |
| # Sample alarm rules. |
| rules: |
| # Rule unique name, must be ended with `_rule`. |
| service_resp_time_rule: |
| indicator-name: service_resp_time |
| op: ">" |
| threshold: 1000 |
| period: 10 |
| count: 3 |
| silence-period: 5 |
| message: Response time of service {name} is more than 1000ms in last 3 minutes. |
| service_sla_rule: |
| # Indicator value need to be long, double or int |
| indicator-name: service_sla |
| op: "<" |
| threshold: 8000 |
| # The length of time to evaluate the metric |
| period: 10 |
| # How many times after the metric match the condition, will trigger alarm |
| count: 2 |
| # How many times of checks, the alarm keeps silence after alarm triggered, default as same as period. |
| silence-period: 3 |
| message: Successful rate of service {name} is lower than 80% in last 2 minutes. |
| service_p90_sla_rule: |
| # Indicator value need to be long, double or int |
| indicator-name: service_p90 |
| op: ">" |
| threshold: 1000 |
| period: 10 |
| count: 3 |
| silence-period: 5 |
| message: 90% response time of service {name} is lower than 1000ms in last 3 minutes |
| service_instance_resp_time_rule: |
| indicator-name: service_instance_resp_time |
| op: ">" |
| threshold: 1000 |
| period: 10 |
| count: 2 |
| silence-period: 5 |
| message: Response time of service instance {name} is more than 1000ms in last 2 minutes. |
| # Active endpoint related metric alarm will cost more memory than service and service instance metric alarm. |
| # Because the number of endpoint is much more than service and instance. |
| # |
| # endpoint_avg_rule: |
| # indicator-name: endpoint_avg |
| # op: ">" |
| # threshold: 1000 |
| # period: 10 |
| # count: 2 |
| # silence-period: 5 |
| # message: Response time of endpoint {name} is more than 1000ms in last 2 minutes. |
| |
| webhooks: |
| # - http://127.0.0.1/notify/ |
| # - http://127.0.0.1/go-wechat/ |
| |