mirror of
https://github.com/go-admin-team/go-admin.git
synced 2026-10-03 13:59:52 +00:00
A second instance pointed at the same database did not divide the work, it
overwrote it. Every instance registered the whole enabled list into its own
cron.Cron, and startup ran `UPDATE sys_job SET entry_id = 0 WHERE entry_id
> 0` across the entire table, so the newest process erased the ids the
previous one wrote and put its own over the top. Neither symptom logged
anything: an enabled job fired once per instance, and stopping one from the
UI removed an entry from the wrong process and answered 200 either way.
A supervisor per tenant now keeps the scheduler in step with the lease.
Holding it is not decided once at startup, because both of the other
answers are wrong for longer than a moment:
- an instance that never got the lease keeps asking, so the death of the
holder does not stop the jobs until somebody restarts a process by hand;
- an instance that holds it stops scheduling as soon as its lease has
lapsed, because a holder still scheduling after the lease has gone
elsewhere is the two-schedulers defect reached from the other side.
A failed renewal is not a lost lease. The scheduler keeps running until the
lease could actually have expired: a database that is briefly unreachable
must not stop the jobs, and cannot have handed them to anyone else, because
nobody else can reach it to take the lease either.
Stopping is no longer arranged by startCrontab. A scheduler now stops for
two different reasons and only the supervisor knows which - and since a
start happens every time the lease is taken, registering a shutdown
callback there would add one per leadership change for the life of the
process, because SetShutdown appends.
sys_job.entry_id keeps its meaning. This does not distribute the jobs and is
not meant to: the HTTP side scales, the scheduler stays single-writer.
Fixes #915.
95 lines
3.2 KiB
Go
95 lines
3.2 KiB
Go
package jobs
|
|
|
|
import (
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/go-admin-team/go-admin-core/v2/sdk"
|
|
"github.com/go-admin-team/go-admin-core/v2/sdk/pkg/cronjob"
|
|
|
|
models2 "go-admin/app/jobs/models"
|
|
)
|
|
|
|
// An instance that starts while another one holds the lease must not
|
|
// register the jobs. This is the whole point: every instance registering the
|
|
// whole enabled list into its own scheduler is what made one job fire once
|
|
// per instance (#915).
|
|
func TestASecondInstanceDoesNotScheduleWhileTheFirstHoldsTheLease(t *testing.T) {
|
|
const tenant = "second-instance"
|
|
db := leaseDB(t)
|
|
sdk.Runtime.SetCrontabByTenant(tenant, cronjob.NewWithSeconds())
|
|
|
|
first := newSupervisor(tenant, db, "instance-a")
|
|
first.tick()
|
|
if !first.holdsLease() {
|
|
t.Fatal("the first instance did not take a free lease")
|
|
}
|
|
|
|
second := newSupervisor(tenant, db, "instance-b")
|
|
second.tick()
|
|
if second.holdsLease() {
|
|
t.Error("a second instance scheduled while the first holds the lease: the job would fire twice per tick")
|
|
}
|
|
}
|
|
|
|
// Losing the lease has to stop the scheduler, not merely stop it from being
|
|
// taken again. A holder that keeps scheduling after its lease has gone to
|
|
// somebody else is two schedulers at once - the defect the lease exists to
|
|
// prevent, reached from the other direction.
|
|
func TestTheSupervisorStopsSchedulingWhenItLosesTheLease(t *testing.T) {
|
|
const tenant = "loses-lease"
|
|
db := leaseDB(t)
|
|
sdk.Runtime.SetCrontabByTenant(tenant, cronjob.NewWithSeconds())
|
|
|
|
holder := newSupervisor(tenant, db, "instance-a")
|
|
holder.tick()
|
|
if !holder.holdsLease() {
|
|
t.Fatal("the supervisor did not take a free lease, so losing it cannot be observed")
|
|
}
|
|
|
|
// What a partition looks like from the database's side: the lease
|
|
// lapsed and somebody else took it while this instance was away.
|
|
expire(t, db)
|
|
successor := &lease{db: db, owner: "instance-b", ttl: time.Minute}
|
|
if held, err := successor.acquire(); err != nil || !held {
|
|
t.Fatalf("the successor could not take the expired lease: held=%v err=%v", held, err)
|
|
}
|
|
|
|
holder.tick()
|
|
|
|
if holder.holdsLease() {
|
|
t.Error("the supervisor kept scheduling after the lease went to another instance")
|
|
}
|
|
if got := readLease(t, db).Owner; got != "instance-b" {
|
|
t.Errorf("owner is %q; the instance that lost the lease wrote over its successor", got)
|
|
}
|
|
}
|
|
|
|
// A database that cannot be reached is not the same as a lease that has been
|
|
// lost. Stopping on the first failed renewal would stop the jobs every time
|
|
// the database blinked - and hand them to nobody, because no other instance
|
|
// can reach it to take the lease either.
|
|
func TestABrieflyUnreachableDatabaseDoesNotStopTheScheduler(t *testing.T) {
|
|
const tenant = "db-blip"
|
|
db := leaseDB(t)
|
|
sdk.Runtime.SetCrontabByTenant(tenant, cronjob.NewWithSeconds())
|
|
|
|
s := newSupervisor(tenant, db, "instance-a")
|
|
s.tick()
|
|
if !s.holdsLease() {
|
|
t.Fatal("the supervisor did not take a free lease")
|
|
}
|
|
|
|
// The table going missing is how an unreachable database presents to
|
|
// acquire: every statement against it returns an error.
|
|
if err := db.Migrator().DropTable(&models2.SysJobLease{}); err != nil {
|
|
t.Fatalf("dropping the lease table: %v", err)
|
|
}
|
|
|
|
s.tick()
|
|
|
|
if !s.holdsLease() {
|
|
t.Error("one failed renewal stopped the scheduler; the lease had not expired yet and nobody else could have taken it")
|
|
}
|
|
}
|