diff --git a/server/cmd/main.go b/server/cmd/main.go index 6976b21..c7a8c5e 100644 --- a/server/cmd/main.go +++ b/server/cmd/main.go @@ -51,10 +51,10 @@ import ( // @name Authorization // @description An API token, sent as "Bearer vt_…". Scoped and optionally expiring. -// @securityDefinitions.apikey esoAuth -// @in header -// @name Authorization -// @description The External Secrets read token, rotated under Settings. It reaches /api/secrets/{group}/values and nothing else. It is a different credential from an API token, and the two must never be substituted for one another. +// @securityDefinitions.apikey esoAuth +// @in header +// @name Authorization +// @description The External Secrets read token, rotated under Settings. It reaches /api/secrets/{group}/values and nothing else. It is a different credential from an API token, and the two must never be substituted for one another. func main() { mongoURI := getEnv("MONGO_URI", "mongodb://localhost:27017") @@ -162,6 +162,10 @@ func runSchemaSetup() { log.Printf("warning: failed to ensure workflow indexes: %v", err) } + if err := services.EnsureMonitorSampleIndexes(); err != nil { + log.Printf("warning: failed to ensure monitor sample indexes: %v", err) + } + if err := services.EnsureVulnIndexes(); err != nil { log.Printf("warning: failed to ensure vuln indexes: %v", err) } diff --git a/server/internal/api/docs/openapi.json b/server/internal/api/docs/openapi.json index 15354c7..d7c8605 100644 --- a/server/internal/api/docs/openapi.json +++ b/server/internal/api/docs/openapi.json @@ -1119,6 +1119,26 @@ }, "type": "object" }, + "models.MonitorSample": { + "properties": { + "at": { + "type": "string" + }, + "instance_id": { + "type": "string" + }, + "latency_ms": { + "type": "integer" + }, + "monitor_id": { + "type": "string" + }, + "up": { + "type": "boolean" + } + }, + "type": "object" + }, "models.MonitorState": { "properties": { "cert_expiry_at": { @@ -4345,6 +4365,77 @@ ] } }, + "/monitors/{id}/samples": { + "get": { + "description": "Raw check results for the last `minutes` minutes, oldest first. Samples expire after 48 hours; use the uptime rollups for longer ranges.", + "parameters": [ + { + "description": "Monitor ID", + "in": "path", + "name": "id", + "required": true, + "schema": { + "type": "string" + } + }, + { + "description": "Window in minutes (default 60, max 2880)", + "in": "query", + "name": "minutes", + "schema": { + "type": "integer" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "items": { + "$ref": "#/components/schemas/models.MonitorSample" + }, + "type": "array" + } + } + }, + "description": "OK" + }, + "404": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/api.ErrorResponse" + } + } + }, + "description": "Not Found" + }, + "500": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/api.ErrorResponse" + } + } + }, + "description": "Internal Server Error" + } + }, + "security": [ + { + "cookieAuth": [] + }, + { + "bearerAuth": [] + } + ], + "summary": "Get a monitor's individual check results", + "tags": [ + "monitors" + ] + } + }, "/monitors/{id}/uptime": { "get": { "description": "Hourly rollups for the last 30 days.", diff --git a/server/internal/api/monitors.go b/server/internal/api/monitors.go index 2adc7a5..c9ce559 100644 --- a/server/internal/api/monitors.go +++ b/server/internal/api/monitors.go @@ -2,6 +2,7 @@ package api import ( "net/http" + "strconv" "time" "gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/auth" @@ -19,6 +20,7 @@ func registerMonitorRoutes(g *gin.RouterGroup) { g.DELETE("/monitors/:id", deleteMonitor) g.GET("/monitors/:id/incidents", getMonitorIncidents) g.GET("/monitors/:id/uptime", getMonitorUptime) + g.GET("/monitors/:id/samples", getMonitorSamples) } // listMonitors godoc @@ -221,6 +223,49 @@ func getMonitorIncidents(c *gin.Context) { c.JSON(http.StatusOK, incidents) } +// getMonitorSamples godoc +// +// @Summary Get a monitor's individual check results +// @Description Raw check results for the last `minutes` minutes, oldest first. Samples expire after 48 hours; use the uptime rollups for longer ranges. +// @Tags monitors +// @Produce json +// @Param id path string true "Monitor ID" +// @Param minutes query int false "Window in minutes (default 60, max 2880)" +// @Success 200 {array} models.MonitorSample +// @Failure 404 {object} ErrorResponse +// @Failure 500 {object} ErrorResponse +// @Security cookieAuth +// @Security bearerAuth +// @Router /monitors/{id}/samples [get] +func getMonitorSamples(c *gin.Context) { + m, err := services.GetMonitor(auth.InstanceID(c), c.Param("id")) + if err != nil { + c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()}) + return + } + if m == nil { + c.JSON(http.StatusNotFound, gin.H{"error": "monitor not found"}) + return + } + // Clamped rather than rejected: the window is a view setting, and the only + // honest answer past the TTL is the shorter window anyway. + minutes := 60 + if raw := c.Query("minutes"); raw != "" { + if n, convErr := strconv.Atoi(raw); convErr == nil && n > 0 { + minutes = n + } + } + if max := int(services.MonitorSampleTTL.Minutes()); minutes > max { + minutes = max + } + samples, err := services.MonitorSamples(auth.InstanceID(c), c.Param("id"), time.Now().Add(-time.Duration(minutes)*time.Minute)) + if err != nil { + c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()}) + return + } + c.JSON(http.StatusOK, samples) +} + // getMonitorUptime godoc // // @Summary Get a monitor's uptime rollups diff --git a/server/internal/api/scopes.go b/server/internal/api/scopes.go index 28779d4..742036f 100644 --- a/server/internal/api/scopes.go +++ b/server/internal/api/scopes.go @@ -100,6 +100,7 @@ var routeScopes = map[string]string{ "DELETE /api/monitors/:id": "monitors:write", "GET /api/monitors/:id/incidents": "monitors:read", "GET /api/monitors/:id/uptime": "monitors:read", + "GET /api/monitors/:id/samples": "monitors:read", // Channel routes, registered by registerChannelRoutes. Channels exist to // serve alerts, so they share the monitors scope rather than getting their diff --git a/server/internal/models/monitor.go b/server/internal/models/monitor.go index 97e70ed..7bf7085 100644 --- a/server/internal/models/monitor.go +++ b/server/internal/models/monitor.go @@ -71,6 +71,21 @@ type Incident struct { Cause string `bson:"cause,omitempty" json:"cause,omitempty"` } +// MonitorSample is one check result, kept only long enough to draw the +// sub-hour views of the history chart. Rollup remains the durable record: a +// sample expires by TTL, a rollup does not. +// +// It carries no message. The failure text is on the incident, and a document +// per check is the one place in this schema where a few bytes multiply by the +// check rate. +type MonitorSample struct { + InstanceID string `bson:"instance_id" json:"instance_id"` + MonitorID string `bson:"monitor_id" json:"monitor_id"` + At time.Time `bson:"at" json:"at"` + Up bool `bson:"up" json:"up"` + LatencyMs int `bson:"latency_ms" json:"latency_ms"` +} + type Rollup struct { InstanceID string `bson:"instance_id" json:"instance_id"` MonitorID string `bson:"monitor_id" json:"monitor_id"` diff --git a/server/internal/services/migrate_instance.go b/server/internal/services/migrate_instance.go index 1149d81..01752fe 100644 --- a/server/internal/services/migrate_instance.go +++ b/server/internal/services/migrate_instance.go @@ -36,6 +36,7 @@ var ScopedCollections = []string{ "monitors", "incidents", "monitor_rollups", + "monitor_samples", "notification_channels", "console_sessions", "audit_logs", diff --git a/server/internal/services/monitors.go b/server/internal/services/monitors.go index 6f06dea..19b8395 100644 --- a/server/internal/services/monitors.go +++ b/server/internal/services/monitors.go @@ -221,6 +221,7 @@ func DeleteMonitor(instanceID, monitorID string) error { } db.Col("incidents").DeleteMany(ctx, bson.M{"monitor_id": monitorID, "instance_id": instanceID}) db.Col("monitor_rollups").DeleteMany(ctx, bson.M{"monitor_id": monitorID, "instance_id": instanceID}) + db.Col("monitor_samples").DeleteMany(ctx, bson.M{"monitor_id": monitorID, "instance_id": instanceID}) return nil } @@ -258,6 +259,31 @@ func UptimeRollups(instanceID, monitorID string, since time.Time) ([]models.Roll return out, nil } +// MaxMonitorSamples bounds one range read. At the 10s floor, 48h is 17,280 +// checks; the chart buckets them anyway, so a cap costs nothing visible and +// stops one monitor pulling a megabyte of JSON per poll. +const MaxMonitorSamples = 6000 + +// MonitorSamples returns individual check results since a point in time, +// oldest first. Samples older than MonitorSampleTTL have expired, so an early +// `since` silently returns a shorter window rather than an error — the caller +// draws the gap. +func MonitorSamples(instanceID, monitorID string, since time.Time) ([]models.MonitorSample, error) { + ctx, cancel := monCtx() + defer cancel() + cur, err := db.Col("monitor_samples").Find(ctx, + bson.M{"monitor_id": monitorID, "instance_id": instanceID, "at": bson.M{"$gte": since}}, + options.Find().SetSort(bson.M{"at": 1}).SetLimit(MaxMonitorSamples)) + if err != nil { + return nil, err + } + var out []models.MonitorSample + if err := cur.All(ctx, &out); err != nil { + return nil, err + } + return out, nil +} + func IngestResult(instanceID, runner, monitorID string, res checker.Result) error { if instanceID == "" { return errors.New("instance id required") @@ -328,6 +354,17 @@ func ingestResult(instanceID, runner, monitorID string, res checker.Result) erro up = 1 } + /* The sample is the same result at full resolution, expiring by TTL. It is + written next to the rollup rather than instead of it: the rollup is what + survives, the sample is what the sub-hour views read. */ + db.Col("monitor_samples").InsertOne(ctx, models.MonitorSample{ + InstanceID: m.InstanceID, + MonitorID: monitorID, + At: now, + Up: res.Up, + LatencyMs: res.LatencyMs, + }) + db.Col("monitor_rollups").UpdateOne(ctx, bson.M{"monitor_id": monitorID, "period_start": bucket}, bson.M{ diff --git a/server/internal/services/monitorsampleindexes.go b/server/internal/services/monitorsampleindexes.go new file mode 100644 index 0000000..e109b6e --- /dev/null +++ b/server/internal/services/monitorsampleindexes.go @@ -0,0 +1,42 @@ +package services + +import ( + "context" + "log" + "time" + + "gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/db" + "go.mongodb.org/mongo-driver/v2/bson" + "go.mongodb.org/mongo-driver/v2/mongo" + "go.mongodb.org/mongo-driver/v2/mongo/options" +) + +// MonitorSampleTTL is how long an individual check result is kept. +// +// It matches the longest range the chart draws from samples rather than the +// longest range it draws at all: 24h and 48h come from the hourly rollups, +// which are permanent. Keeping samples past the window that reads them would +// only grow the collection. +const MonitorSampleTTL = 48 * time.Hour + +// EnsureMonitorSampleIndexes declares the sample range index and its TTL. +// +// Warn rather than fatal, like the other history indexes — but note the TTL is +// not an optimisation: without it nothing ever removes a sample, and the +// collection grows at the fleet's total check rate forever. A boot that logs +// this warning needs following up. +func EnsureMonitorSampleIndexes() error { + ctx := context.Background() + + idx := []mongo.IndexModel{ + // Every read is a range scan over this key. + {Keys: bson.D{{Key: "instance_id", Value: 1}, {Key: "monitor_id", Value: 1}, {Key: "at", Value: 1}}}, + // Expiry is Mongo's job: a sweeper would be another leader-scoped loop + // doing what the server already does for free. + {Keys: bson.D{{Key: "at", Value: 1}}, Options: options.Index().SetExpireAfterSeconds(int32(MonitorSampleTTL.Seconds()))}, + } + if _, err := db.Col("monitor_samples").Indexes().CreateMany(ctx, idx); err != nil { + log.Printf("warning: monitor_samples indexes: %v", err) + } + return nil +} diff --git a/web/app/(app)/monitors/[id]/page.tsx b/web/app/(app)/monitors/[id]/page.tsx index 8da3ee9..b09bd1e 100644 --- a/web/app/(app)/monitors/[id]/page.tsx +++ b/web/app/(app)/monitors/[id]/page.tsx @@ -10,6 +10,7 @@ import { Slot, StatusChip, avgLatency, + buildSampleSlots, buildSlots, displayStatus, formatDuration, @@ -31,6 +32,63 @@ import { const CHART_W = 480; const CHART_H = 158; +/* + * The ranges. 24h and 48h are drawn from the hourly rollups, which are the + * permanent record; anything shorter than an hour cannot be, so the three + * short ranges read individual check results instead. Those expire after 48 + * hours, which is why no range longer than that is offered from samples. + * + * Bucket sizes are chosen to land near 50-70 bars, so the tape has the same + * texture whichever range is selected. + */ +interface Range { + label: string; + minutes: number; + /** "rollups" is hourly and permanent; "samples" is per check and expires. */ + source: "rollups" | "samples"; + bucketMs: number; +} + +const RANGES: Range[] = [ + { label: "48h", minutes: 48 * 60, source: "rollups", bucketMs: 3600_000 }, + { label: "24h", minutes: 24 * 60, source: "rollups", bucketMs: 3600_000 }, + { label: "12h", minutes: 12 * 60, source: "samples", bucketMs: 600_000 }, + { label: "8h", minutes: 8 * 60, source: "samples", bucketMs: 600_000 }, + { label: "1h", minutes: 60, source: "samples", bucketMs: 60_000 }, +]; + +function rangeTitle(r: Range): string { + const hours = r.minutes / 60; + return hours === 1 ? "Last hour" : `Last ${hours} hours`; +} + +function bucketLabel(bucketMs: number): string { + if (bucketMs >= 3600_000) return "1 hour per bar"; + return `${Math.round(bucketMs / 60_000)} min per bar`; +} + +function RangePicker({ value, onChange }: { value: Range; onChange: (r: Range) => void }) { + return ( +