diff --git a/README.md b/README.md index 7d78e53c62a5a59190b0609bdf6b5ad5193f1e4f..328f9f8ee91a925f5164b8a8d7549bd3ecf49018 100644 --- a/README.md +++ b/README.md @@ -10,19 +10,19 @@ v3.x的版本和v2.x差别巨大,没办法平滑迁移,可以继续使用 [v 用户资源中心: -![](https://s3-gz01.didistatic.com/n9e-pub/image/snapshot/rdb.png) +![用户资源中心截图](https://s3-gz01.didistatic.com/n9e-pub/image/snapshot/rdb.png) 资产管理中心: -![](https://s3-gz01.didistatic.com/n9e-pub/image/snapshot/ams.png) +![资产管理中心截图](https://s3-gz01.didistatic.com/n9e-pub/image/snapshot/ams.png) 任务执行中心: -![](https://s3-gz01.didistatic.com/n9e-pub/image/snapshot/job.png) +![任务执行中心截图](https://s3-gz01.didistatic.com/n9e-pub/image/snapshot/job.png) 监控告警中心: -![](https://s3-gz01.didistatic.com/n9e-pub/image/snapshot/mon.png) +![监控告警中心截图](https://s3-gz01.didistatic.com/n9e-pub/image/snapshot/mon.png) # 安装步骤 @@ -114,6 +114,8 @@ cd /home/n9e setenforce 0 ``` +上面安装步骤如果走完了仍然没有搭建起来,你可能需要 [使用Docker安装](dockerfiles/README.md) 或者 [查看视频教程](https://mp.weixin.qq.com/s/OAEQ-ec-QM74U0SGoVCXkg) + # 子系统简介 夜莺拆成了四个子系统,分别是:用户资源中心(RDB)、资产管理系统(AMS)、任务执行中心(JOB)、监控告警系统(MON)。下面分别介绍一下这几个子系统的设计初衷 @@ -140,4 +142,9 @@ setenforce 0 这块核心逻辑和v2版本差别不大,监控指标分成了设备相关指标和设备无关指标,因为有些自定义监控数据的场景,endpoint不好定义,或者endpoint经常变化,这种就可以使用设备无关指标的方式来处理。监控大盘做了优化,引入了更多类型的图表,但夜莺毕竟是个metrics监控系统,处理的是数值型时序数据,所以,最有用的图表其实就是折线图,其他类型图表,看看就好,场景较少。夜莺也可以对接Grafana,有个专门的[DataSource插件](https://github.com/n9e/grafana-n9e-datasource),Grafana会更炫酷一些,只是,在数据量大的时候性能较差。 +# 系统架构 + +![n9e系统架构图](https://s3-gz01.didistatic.com/n9e-pub/image/n9e-v3-arch.png) + +监控部分的架构和之前没有差别,collector揉进了一些命令执行的能力,所以改了个名字叫agent。引入了三个新组件:rdb、ams、job,rdb是用户资源中心,ams是资产管理系统,job是任务执行中心。agent除了上报监控数据给transfer,还会上报本机信息给ams,注册本机信息到资产管理系统,另外就是与job模块交互,拉取要执行的任务,上报任务执行结果。 diff --git a/etc/json/stra.json b/etc/json/stra.json new file mode 100644 index 0000000000000000000000000000000000000000..034f1188aad523ce65bf68f8f83a0061f91d11f6 --- /dev/null +++ b/etc/json/stra.json @@ -0,0 +1,410 @@ +[ + { + "name": "内存利用率大于75%", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 2, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "mem.bytes.used.percent", + "params": [], + "threshold": 75 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "机器loadavg大于16", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 2, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "cpu.loadavg.1", + "params": [], + "threshold": 16 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "某磁盘无法正常读写", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 1, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "disk.rw.error", + "params": [], + "threshold": 0 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "监控agent失联", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 1, + "exprs": [ + { + "eopt": "=", + "func": "nodata", + "metric": "proc.agent.alive", + "params": [], + "threshold": 0 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "磁盘利用率达到85%", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 3, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "disk.bytes.used.percent", + "params": [], + "threshold": 85 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "磁盘利用率达到88%", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 2, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "disk.bytes.used.percent", + "params": [], + "threshold": 88 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "磁盘利用率达到92%", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 1, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "disk.bytes.used.percent", + "params": [], + "threshold": 92 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "端口挂了", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 2, + "exprs": [ + { + "eopt": "!=", + "func": "all", + "metric": "proc.port.listen", + "params": [], + "threshold": 1 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "网卡入方向丢包", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 2, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "net.in.dropped", + "params": [], + "threshold": 3 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "网卡出方向丢包", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 2, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "net.out.dropped", + "params": [], + "threshold": 3 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "进程总数超过3000", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 1, + "exprs": [ + { + "eopt": ">", + "func": "all", + "metric": "sys.ps.process.total", + "params": [], + "threshold": 3000 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + }, + { + "name": "进程挂了", + "category": 1, + "alert_dur": 60, + "recovery_dur": 0, + "recovery_notify": 1, + "enable_stime": "00:00", + "enable_etime": "23:59", + "priority": 2, + "exprs": [ + { + "eopt": "<", + "func": "all", + "metric": "proc.num", + "params": [], + "threshold": 1 + } + ], + "tags": [], + "enable_days_of_week": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6 + ], + "converge": [ + 36000, + 1 + ], + "endpoints": null + } +] \ No newline at end of file