Files
mrhid6 f552006318
Agent Release / build (push) Successful in 3m36s
Agent Release / msi (push) Successful in 2m30s
feat: flag Ubuntu phased updates and leave them out of pending counts
apt lists phased updates as upgradable, but an upgrade defers them until the
host is selected. A simulated upgrade names them; they are reported with the
phased flag (vantage-shared v0.6.0) and not counted in pending_after.
2026-09-15 14:55:33 +00:00

136 lines
5.3 KiB
Go

package agentsync
import (
"errors"
"log"
"time"
"gitea.hostxtra.co.uk/vantage/vantage-agent/internal/config"
grpcclient "gitea.hostxtra.co.uk/vantage/vantage-agent/internal/grpc"
"gitea.hostxtra.co.uk/vantage/vantage-agent/internal/updates"
"gitea.hostxtra.co.uk/vantage/vantage-shared/grpc/pb"
)
// minRebootLeeway is the least time that must remain in the window for the
// agent to start a reboot. A reboot that lands after the window closes is the
// outage the window existed to prevent.
const minRebootLeeway = 5 * time.Minute
// shouldReboot is the whole reboot decision. The agent reboots a host only
// when the command asked, the OS reports a reboot is owed, and the window
// still has room for it. A command with no deadline is a manual run, which
// never reboots.
func shouldReboot(requested, owed bool, now, deadline time.Time) bool {
if !requested || !owed || deadline.IsZero() {
return false
}
return deadline.Sub(now) >= minRebootLeeway
}
func handleApplyUpdates(send func(*pb.AgentMessage) error, cfg *config.Config, cmd *pb.ServerCommand) {
c := cmd.ApplyUpdates
var deadline time.Time
if c.DeadlineUnix > 0 {
deadline = time.Unix(c.DeadlineUnix, 0)
}
log.Printf("applying OS updates (cmd=%s scope=%q reboot=%v)", cmd.CommandId, c.Scope, c.RebootIfRequired)
res, err := updates.Apply(updates.ApplyOptions{SecurityOnly: c.Scope == pb.PatchScopeSecurity, Deadline: deadline})
pr := &pb.PatchResult{CommandId: cmd.CommandId, OutputTail: res.Output, PendingAfter: -1}
switch {
case errors.Is(err, updates.ErrBusy):
pr.Status, pr.Message = pb.PatchStatusBusy, err.Error()
case err != nil:
pr.Status, pr.Message = pb.PatchStatusFailed, err.Error()
case res.Unsupported:
pr.Status, pr.Message = pb.PatchStatusUnsupported, res.Reason
default:
pr.Status = pb.PatchStatusOK
}
// Refresh the pending list whether the run succeeded or not, so the counts
// the operator sees are this host's real state. A busy refusal changed
// nothing, and the run in progress will report for itself.
if pr.Status != pb.PatchStatusBusy {
pr.PendingAfter = int32(reportPendingUpdates(cfg))
}
pr.RebootRequired = updates.RebootRequired()
pr.Rebooting = pr.Status == pb.PatchStatusOK && shouldReboot(c.RebootIfRequired, pr.RebootRequired, time.Now(), deadline)
msg := &pb.AgentMessage{ServerId: cfg.ServerID, AgentToken: cfg.AgentToken, PatchResult: pr}
if err := send(msg); err != nil {
log.Printf("send patch result (cmd=%s): %v; keeping it for the next command stream", cmd.CommandId, err)
// A reboot nobody was told about looks like a crash. Without a
// delivered result, do not reboot: the queued result retakes the
// decision when it is finally sent.
if pendingResults.push(pendingResult{msg: msg, requested: c.RebootIfRequired, deadline: deadline}) {
log.Printf("patch result queue full: dropped the oldest undelivered result")
}
return
}
log.Printf("patch result sent (cmd=%s status=%s pending_after=%d rebooting=%v)", cmd.CommandId, pr.Status, pr.PendingAfter, pr.Rebooting)
if pr.Rebooting {
if err := updates.ScheduleReboot(); err != nil {
log.Printf("schedule reboot (cmd=%s): %v", cmd.CommandId, err)
}
}
}
// pendingResults holds PatchResults whose send failed, typically because the
// command stream reconnected while a patch ran. The server guards result
// writes on status and command id, so a late result is safe to deliver.
var pendingResults = &resultQueue{max: maxPendingResults}
// flushPendingResults delivers queued results on a newly established stream.
// A queued result that announced a reboot has that decision taken again now:
// time has passed, so the window may be too close to its end, or the reboot
// may no longer be owed. Rebooting is cleared before sending when it no
// longer holds, and the host reboots only once the result is delivered.
func flushPendingResults(send func(*pb.AgentMessage) error) {
pendingResults.flush(func(p pendingResult) error {
pr := p.msg.PatchResult
if pr.Rebooting {
owed := updates.RebootRequired()
pr.RebootRequired = owed
if !shouldReboot(p.requested, owed, time.Now(), p.deadline) {
pr.Rebooting = false
}
}
if err := send(p.msg); err != nil {
log.Printf("send queued patch result (cmd=%s): %v", pr.CommandId, err)
return err
}
log.Printf("queued patch result sent (cmd=%s status=%s rebooting=%v)", pr.CommandId, pr.Status, pr.Rebooting)
if pr.Rebooting {
if err := updates.ScheduleReboot(); err != nil {
log.Printf("schedule reboot (cmd=%s): %v", pr.CommandId, err)
}
}
return nil
})
}
// reportPendingUpdates re-checks pending updates and reports them, returning
// the count, or -1 if either step failed.
func reportPendingUpdates(cfg *config.Config) int {
pkgs, err := updates.CheckAvailable()
if err != nil {
log.Printf("post-apply update check: %v", err)
return -1
}
list := make([]pb.PackageUpdate, len(pkgs))
for i, p := range pkgs {
list[i] = pb.PackageUpdate{Name: p.Name, CurrentVersion: p.CurrentVersion, NewVersion: p.NewVersion, Phased: p.Phased}
}
client, err := grpcclient.New(cfg.ServerURL, cfg.TLS)
if err != nil {
log.Printf("post-apply report dial: %v", err)
return updates.CountInstallable(pkgs)
}
defer client.Close()
if err := client.ReportUpdates(cfg.ServerID, cfg.AgentToken, list); err != nil {
log.Printf("post-apply ReportUpdates: %v", err)
}
return len(pkgs)
}