mirror of
https://github.com/go-admin-team/go-admin.git
synced 2026-10-01 21:27:14 +00:00
The manifest in this repository had no probes at all. A pod was sent traffic as soon as its container was running, whether or not the database it needs was reachable, and it was stopped with whatever grace period Kubernetes defaults to rather than one chosen against what this process actually spends shutting down. It now mounts both probes, at the endpoint that answers each question: readiness at /ready, which fails while a dependency is unreachable, and liveness at /health, which is a bare 200 because restarting a process whose database is down turns one outage into a crash loop. Both skip the rate limiter, which is why that had to land first. timeoutSeconds is 3, not the default 1. The handler allows its checks two seconds, so at the default a database answering in 1.2s would be recorded as a failed check while the handler was returning 200 - the probe would be failing on the orchestrator's stopwatch, not on its own. The comment beside that constant said the constraint was the polling period; the constraint is the per-check timeout, and it is now written down correctly. terminationGracePeriodSeconds is 30, against a shipped budget of 0 + 5 + 3. Raising drain means raising this too, in the same commit; the check that notices when somebody does not arrives two commits from here. replicas stays at 1, and the comment says why that makes the drain window worth nothing: there is nowhere to send the traffic this pod stops taking. Raising it needs one more change than the number - the volume is shared by every replica and the log path lives on it, so a second pod would append to the same rotating file. The reason not to raise it is not the one the review assumed: the claim was that the PVC is ReadWriteOnce, and it is not, it is ReadWriteMany on nfs-csi. There is no preStop hook. How long one should sleep depends on how fast the thing in front removes this instance, which the repository cannot know, and a manifest carrying both a preStop sleep and a drain window is the double-counting trap - the budget would be spent twice and the start-up line would report half of it.
83 lines
2.7 KiB
Go
83 lines
2.7 KiB
Go
package router
|
|
|
|
import (
|
|
"context"
|
|
"net/http"
|
|
"time"
|
|
|
|
"github.com/gin-gonic/gin"
|
|
"github.com/go-admin-team/go-admin-core/v2/tools/transfer"
|
|
"github.com/prometheus/client_golang/prometheus/promhttp"
|
|
|
|
"go-admin/common/health"
|
|
)
|
|
|
|
func init() {
|
|
routerNoCheckRole = append(routerNoCheckRole, RegisterMonitorRouter)
|
|
}
|
|
|
|
// readyTimeout bounds the whole probe. What constrains it is the orchestrator's
|
|
// per-check timeout rather than its polling period: Kubernetes allows a probe
|
|
// one second by default, so a dependency that answers in 1.2s is recorded as a
|
|
// failed check however promptly this handler returns. A manifest that mounts
|
|
// this probe has to raise timeoutSeconds above this value, and
|
|
// scripts/k8s/deploy.yml does.
|
|
const readyTimeout = 2 * time.Second
|
|
|
|
// HealthPath and ReadyPath are the two probe routes, relative to APIPrefix.
|
|
//
|
|
// Exported for the same reason as the prefix: the rate limiter has to be told
|
|
// to skip them, and it is installed in a package that cannot import this one.
|
|
const (
|
|
HealthPath = "/health"
|
|
ReadyPath = "/ready"
|
|
)
|
|
|
|
// RegisterMonitorRouter mounts the metrics endpoint and the two probes on v1.
|
|
//
|
|
// Exported so that a test can put the real probes on a server of its own. The
|
|
// alternative - a test that re-implements the handler it means to check - is
|
|
// how a probe comes to be asserted against a copy of itself.
|
|
//
|
|
// 无需认证的路由代码
|
|
func RegisterMonitorRouter(v1 *gin.RouterGroup) {
|
|
v1.GET("/metrics", transfer.Handler(promhttp.Handler()))
|
|
|
|
// 健康检查(存活)
|
|
//
|
|
// Stays a bare 200 on purpose. This is the answer to "should I restart
|
|
// you", and a process whose database is unreachable does not want
|
|
// restarting - that turns one outage into a crash loop and throws away the
|
|
// connection pool, the cache and every in-flight request along the way.
|
|
v1.GET(HealthPath, func(c *gin.Context) {
|
|
c.Status(http.StatusOK)
|
|
})
|
|
|
|
// 就绪检查
|
|
//
|
|
// The answer to "should I send you requests". It fails while a dependency
|
|
// is unreachable, and from the moment shutdown begins - for as long as
|
|
// extend.shutdown.drain says, which is zero unless it is configured. The
|
|
// package comment in common/health says what that window is worth, and to
|
|
// whom.
|
|
v1.GET(ReadyPath, func(c *gin.Context) {
|
|
if health.Draining() {
|
|
c.JSON(http.StatusServiceUnavailable, gin.H{
|
|
"status": "draining",
|
|
"checks": []health.Check{},
|
|
})
|
|
return
|
|
}
|
|
ctx, cancel := context.WithTimeout(c.Request.Context(), readyTimeout)
|
|
defer cancel()
|
|
|
|
checks := health.Ready(ctx)
|
|
status := http.StatusOK
|
|
if !health.Healthy(checks) {
|
|
status = http.StatusServiceUnavailable
|
|
}
|
|
c.JSON(status, gin.H{"status": http.StatusText(status), "checks": checks})
|
|
})
|
|
|
|
}
|