Make endpoint alarm setting not active in default (#1998)

This commit is contained in:
吴晟 Wu Sheng 2018-12-03 22:58:23 +08:00 committed by GitHub
parent 1ea977e50a
commit fc79182c31
No known key found for this signature in database
GPG Key ID: 4AEE18F83AFDEB23
2 changed files with 55 additions and 26 deletions

View File

@ -17,28 +17,54 @@
# Sample alarm rules.
rules:
# Rule unique name, must be ended with `_rule`.
# endpoint_percent_rule:
# # Indicator value need to be long, double or int
# indicator-name: endpoint_percent
# threshold: 75
# op: <
# # The length of time to evaluate the metric
# period: 10
# # How many times after the metric match the condition, will trigger alarm
# count: 3
# # How many times of checks, the alarm keeps silence after alarm triggered, default as same as period.
# silence-period: 10
# message: Successful rate of endpoint {name} is lower than 75%
service_resp_time_rule:
indicator-name: service_resp_time
# [Optional] Default, match all services in this indicator
include-names:
- dubbox-provider
- dubbox-consumer
threshold: 1000
op: ">"
threshold: 1000
period: 10
count: 1
count: 3
silence-period: 5
message: Response time of service {name} is more than 1000ms in last 3 minutes.
service_sla_rule:
# Indicator value need to be long, double or int
indicator-name: service_sla
op: "<"
threshold: 8000
# The length of time to evaluate the metric
period: 10
# How many times after the metric match the condition, will trigger alarm
count: 2
# How many times of checks, the alarm keeps silence after alarm triggered, default as same as period.
silence-period: 3
message: Successful rate of service {name} is lower than 80% in last 2 minutes.
service_p90_sla_rule:
# Indicator value need to be long, double or int
indicator-name: service_p90
op: ">"
threshold: 1000
period: 10
count: 3
silence-period: 5
message: 90% response time of service {name} is lower than 1000ms in last 3 minutes
service_instance_resp_time_rule:
indicator-name: service_instance_resp_time
op: ">"
threshold: 1000
period: 10
count: 2
silence-period: 5
message: Response time of service instance {name} is more than 1000ms in last 2 minutes.
# Active endpoint related metric alarm will cost more memory than service and service instance metric alarm.
# Because the number of endpoint is much more than service and instance.
#
# endpoint_avg_rule:
# indicator-name: endpoint_avg
# op: ">"
# threshold: 1000
# period: 10
# count: 2
# silence-period: 5
# message: Response time of endpoint {name} is more than 1000ms in last 2 minutes.
webhooks:
# - http://127.0.0.1/notify/

View File

@ -54,14 +54,17 @@ rules:
count: 2
silence-period: 5
message: Response time of service instance {name} is more than 1000ms in last 2 minutes.
endpoint_avg_rule:
indicator-name: endpoint_avg
op: ">"
threshold: 1000
period: 10
count: 2
silence-period: 5
message: Response time of endpoint {name} is more than 1000ms in last 2 minutes.
# Active endpoint related metric alarm will cost more memory than service and service instance metric alarm.
# Because the number of endpoint is much more than service and instance.
#
# endpoint_avg_rule:
# indicator-name: endpoint_avg
# op: ">"
# threshold: 1000
# period: 10
# count: 2
# silence-period: 5
# message: Response time of endpoint {name} is more than 1000ms in last 2 minutes.
webhooks:
# - http://127.0.0.1/notify/