You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
203 lines
5.8 KiB
203 lines
5.8 KiB
package console
|
|
|
|
// ============================ 第三方服务巡检域 (svcinspect) ============================
|
|
//
|
|
// 读 svc_health(巡检器 inspect.go 写入)供后台展示,外加手动触发巡检。
|
|
// 三个接口分工:
|
|
// api_getsvchealth 巡检页:全量明细 + 汇总
|
|
// api_runsvcinspect 巡检页:立即巡检(全量或单个服务)
|
|
// api_getsvcalerts 顶栏通知铃铛:只回异常项,尽量小
|
|
//
|
|
// 巡检结果不含任何凭据明文(探针只把服务商的错误描述写进 probe_msg),可直接下发前端。
|
|
|
|
import (
|
|
"sort"
|
|
"strings"
|
|
|
|
"yunyan/comm"
|
|
"yunyan/lego/sys/postgres"
|
|
"yunyan/pb"
|
|
|
|
"github.com/gin-gonic/gin"
|
|
)
|
|
|
|
// requireInspectAccess 巡检结果只对内部角色开放(超管/管理员/运营)。
|
|
// 品牌商(3)/渠道商(5) 是外部角色,「服务巡检」菜单本就不在它们的可见上限内,
|
|
// 接口这一层也一并挡住,免得绕过前端直接读到平台第三方服务的健康状况。
|
|
func (this *serverComp) requireInspectAccess(c *gin.Context) bool {
|
|
switch currentIdentity(c) {
|
|
case pb.Identity_Admin, pb.Identity_Manager, pb.Identity_Operator:
|
|
return true
|
|
}
|
|
writeErr(c, pb.ErrorCode_InsufficientPermissions, "无权查看服务巡检")
|
|
return false
|
|
}
|
|
|
|
// healthSummary 各状态计数,前端顶部汇总卡片直接用。
|
|
type healthSummary struct {
|
|
Total int `json:"total"`
|
|
Ok int `json:"ok"`
|
|
Warn int `json:"warn"`
|
|
Error int `json:"error"`
|
|
Skip int `json:"skip"`
|
|
}
|
|
|
|
// getSvcHealth 列出巡检结果。app_name 传 "__all__" 或省略即不限作用域(巡检覆盖全部作用域,
|
|
// 而 app_name='' 的全局默认服务对所有应用生效,默认展示全部更符合运维视角)。
|
|
func (this *serverComp) getSvcHealth(c *gin.Context) {
|
|
if !this.requireInspectAccess(c) {
|
|
return
|
|
}
|
|
var req struct {
|
|
AppName string `json:"app_name"`
|
|
OnlyAbnormal bool `json:"only_abnormal"`
|
|
}
|
|
_ = c.ShouldBindJSON(&req)
|
|
|
|
rows := make([]*comm.SvcHealth, 0)
|
|
where, args := "", []interface{}{}
|
|
if req.AppName != "" && req.AppName != "__all__" {
|
|
where, args = "app_name=?", []interface{}{req.AppName}
|
|
}
|
|
if err := postgres.Find(comm.TableSvcHealth, &rows, where, args...); err != nil {
|
|
writeErr(c, pb.ErrorCode_DBError, err.Error())
|
|
return
|
|
}
|
|
|
|
sum := healthSummary{}
|
|
items := make([]*comm.SvcHealth, 0, len(rows))
|
|
for _, r := range rows {
|
|
sum.Total++
|
|
switch r.Status {
|
|
case comm.HealthOK:
|
|
sum.Ok++
|
|
case comm.HealthWarn:
|
|
sum.Warn++
|
|
case comm.HealthError:
|
|
sum.Error++
|
|
case comm.HealthSkip:
|
|
sum.Skip++
|
|
}
|
|
if req.OnlyAbnormal && r.Status != comm.HealthError && r.Status != comm.HealthWarn {
|
|
continue
|
|
}
|
|
items = append(items, r)
|
|
}
|
|
sortHealth(items)
|
|
writeOK(c, gin.H{
|
|
"items": items,
|
|
"summary": sum,
|
|
"last_run": this.inspectLastRun(),
|
|
"running": this.inspectRunning(),
|
|
})
|
|
}
|
|
|
|
// runSvcInspect 手动触发巡检。svc_id 非空时只重测该服务(含其全部区域覆盖),否则全量。
|
|
// 全量巡检可能持续数十秒,但仍同步执行——运营点「立即巡检」就是要看到这一轮的结果,
|
|
// 异步返回会让前端只能靠轮询猜什么时候好了。
|
|
func (this *serverComp) runSvcInspect(c *gin.Context) {
|
|
if !this.requireInspectAccess(c) {
|
|
return
|
|
}
|
|
var req struct {
|
|
AppName string `json:"app_name"`
|
|
SvcId string `json:"svc_id"`
|
|
}
|
|
_ = c.ShouldBindJSON(&req)
|
|
|
|
insp := this.module.inspect
|
|
if insp == nil {
|
|
writeErr(c, pb.ErrorCode_SystemError, "巡检组件未就绪")
|
|
return
|
|
}
|
|
if strings.TrimSpace(req.SvcId) != "" {
|
|
n := insp.RunOne(req.AppName, strings.TrimSpace(req.SvcId))
|
|
writeOK(c, gin.H{"ok": true, "count": n})
|
|
return
|
|
}
|
|
n := insp.RunAll("manual")
|
|
if n < 0 {
|
|
writeErr(c, pb.ErrorCode_ReqParameterError, "已有一轮巡检正在进行,请稍候再试")
|
|
return
|
|
}
|
|
writeOK(c, gin.H{"ok": true, "count": n})
|
|
}
|
|
|
|
// getSvcAlerts 顶栏铃铛的数据源:只回异常项(error 在前,warn 在后),并做数量上限,
|
|
// 避免配置量大时把整张巡检表塞进通知面板。
|
|
func (this *serverComp) getSvcAlerts(c *gin.Context) {
|
|
if !this.requireInspectAccess(c) {
|
|
return
|
|
}
|
|
var req struct {
|
|
Limit int `json:"limit"`
|
|
}
|
|
_ = c.ShouldBindJSON(&req)
|
|
if req.Limit <= 0 || req.Limit > 50 {
|
|
req.Limit = 20
|
|
}
|
|
|
|
rows := make([]*comm.SvcHealth, 0)
|
|
if err := postgres.Find(comm.TableSvcHealth, &rows, "status IN (?)",
|
|
[]string{comm.HealthError, comm.HealthWarn}); err != nil {
|
|
writeErr(c, pb.ErrorCode_DBError, err.Error())
|
|
return
|
|
}
|
|
errCnt, warnCnt := 0, 0
|
|
for _, r := range rows {
|
|
if r.Status == comm.HealthError {
|
|
errCnt++
|
|
} else {
|
|
warnCnt++
|
|
}
|
|
}
|
|
sortHealth(rows)
|
|
if len(rows) > req.Limit {
|
|
rows = rows[:req.Limit]
|
|
}
|
|
writeOK(c, gin.H{
|
|
"alerts": rows,
|
|
"error_count": errCnt,
|
|
"warn_count": warnCnt,
|
|
"last_run": this.inspectLastRun(),
|
|
})
|
|
}
|
|
|
|
// sortHealth 展示排序:先按严重程度(error > warn > ok > skip),同级按连续失败次数降序
|
|
// (一直坏着的排在偶发失败前面),再按服务名稳定排序。
|
|
func sortHealth(rows []*comm.SvcHealth) {
|
|
rank := map[string]int{comm.HealthError: 0, comm.HealthWarn: 1, comm.HealthOK: 2, comm.HealthSkip: 3}
|
|
sort.SliceStable(rows, func(i, j int) bool {
|
|
ri, rj := rank[rows[i].Status], rank[rows[j].Status]
|
|
if ri != rj {
|
|
return ri < rj
|
|
}
|
|
if rows[i].FailStreak != rows[j].FailStreak {
|
|
return rows[i].FailStreak > rows[j].FailStreak
|
|
}
|
|
if rows[i].Name != rows[j].Name {
|
|
return rows[i].Name < rows[j].Name
|
|
}
|
|
return rows[i].Region < rows[j].Region
|
|
})
|
|
}
|
|
|
|
func (this *serverComp) inspectLastRun() int64 {
|
|
insp := this.module.inspect
|
|
if insp == nil {
|
|
return 0
|
|
}
|
|
insp.mu.Lock()
|
|
defer insp.mu.Unlock()
|
|
return insp.lastRun
|
|
}
|
|
|
|
func (this *serverComp) inspectRunning() bool {
|
|
insp := this.module.inspect
|
|
if insp == nil {
|
|
return false
|
|
}
|
|
insp.mu.Lock()
|
|
defer insp.mu.Unlock()
|
|
return insp.running
|
|
}
|
|
|