Compare commits
86
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7b13944c24 | ||
|
|
e7384d334a | ||
|
|
be299845ca | ||
|
|
d83061786c | ||
|
|
b28a2263bb | ||
|
|
b6f45390c4 | ||
|
|
2e4bc687d4 | ||
|
|
d328d3aaca | ||
|
|
89cc524f3f | ||
|
|
f0f2600bed | ||
|
|
884fd189fb | ||
|
|
e08cc2f928 | ||
|
|
3ce963b0cd | ||
|
|
461a79277d | ||
|
|
d9f7fa6993 | ||
|
|
060fa64339 | ||
|
|
30d83c4c32 | ||
|
|
a918b1bdc1 | ||
|
|
66b1a041ba | ||
|
|
d55ed2b19a | ||
|
|
8135c8d781 | ||
|
|
577b060b8a | ||
|
|
1028a2e43a | ||
|
|
eba93a812e | ||
|
|
83b7256b60 | ||
|
|
9d17f539b5 | ||
|
|
28f746c7e2 | ||
|
|
9d218cb19f | ||
|
|
f9ef9c4929 | ||
|
|
049e005873 | ||
|
|
c440b59b93 | ||
|
|
3e4865884c | ||
|
|
270d55e6a6 | ||
|
|
7a0a1953f6 | ||
|
|
3e4ccc9720 | ||
|
|
e5947489e4 | ||
|
|
0a7a10aeed | ||
|
|
28b813ba64 | ||
|
|
68160dc681 | ||
|
|
32e7420d89 | ||
|
|
7e1d67dba4 | ||
|
|
da1dc90ac5 | ||
|
|
72c9492223 | ||
|
|
fa67d839cd | ||
|
|
1452928b75 | ||
|
|
bf10023f35 | ||
|
|
f4f41e400b | ||
|
|
3abbdc41d6 | ||
|
|
6ba54f690c | ||
|
|
a3c6b2a305 | ||
|
|
21a2d077d8 | ||
|
|
6263c7e16f | ||
|
|
161835802d | ||
|
|
6f998ff506 | ||
|
|
d192589790 | ||
|
|
21c2bb2646 | ||
|
|
9c0bbd13dd | ||
|
|
1c15961309 | ||
|
|
383b763a66 | ||
|
|
3e99a9df33 | ||
|
|
f1b6f90345 | ||
|
|
2a660697c5 | ||
|
|
0c08dda635 | ||
|
|
22b99ff895 | ||
|
|
2fab784ba7 | ||
|
|
83cdf92575 | ||
|
|
aa1c8e4aa1 | ||
|
|
ac61015cc0 | ||
|
|
a0fbf5b9ba | ||
|
|
ddf0814803 | ||
|
|
2bc15648b7 | ||
|
|
7f348b7b2b | ||
|
|
a71fd9a9f4 | ||
|
|
36848a519a | ||
|
|
ea6d0b969c | ||
|
|
4ba9983cf8 | ||
|
|
8caaa4540f | ||
|
|
42f0cb67cb | ||
|
|
1dfb3cc28c | ||
|
|
b51e87477e | ||
|
|
05c0ad43d2 | ||
|
|
4a07af049c | ||
|
|
057c193e26 | ||
|
|
38a731f269 | ||
|
|
06d69590fc | ||
|
|
6adee810dc |
@@ -76,6 +76,13 @@ jobs:
|
||||
fi
|
||||
echo "ok: reaper configured in cloud mode only"
|
||||
|
||||
- name: Render with backups enabled
|
||||
run: |
|
||||
helm template test "$CHART_DIR" \
|
||||
--set backup.enabled=true \
|
||||
--set backup.image=gitea.hostxtra.co.uk/mrhid6/vantage/vantagectl:latest \
|
||||
--set backup.pvcName=vantage-backups > /dev/null
|
||||
|
||||
- name: Render against external Redis and MongoDB
|
||||
run: |
|
||||
helm template test "$CHART_DIR" \
|
||||
@@ -140,10 +147,21 @@ jobs:
|
||||
--set ingress.enabled=true \
|
||||
--set ingress.web.host=vantage.example.com \
|
||||
--set server.env.grpcHost=agents.example.com:443
|
||||
refuses "an ingress that leaves /api unrouted" \
|
||||
--set ingress.enabled=true \
|
||||
--set ingress.web.host=vantage.example.com \
|
||||
--set ingress.grpc.enabled=false \
|
||||
--set ingress.api.enabled=false
|
||||
refuses "gRPC ingress while grpcHost is still in-cluster" \
|
||||
--set ingress.enabled=true \
|
||||
--set ingress.web.host=vantage.example.com \
|
||||
--set ingress.grpc.host=agents.example.com
|
||||
refuses "backup enabled with no pvcName" \
|
||||
--set backup.enabled=true \
|
||||
--set backup.image=gitea.hostxtra.co.uk/mrhid6/vantage/vantagectl:latest
|
||||
refuses "backup enabled with no image" \
|
||||
--set backup.enabled=true \
|
||||
--set backup.pvcName=vantage-backups
|
||||
|
||||
- name: Read the chart version
|
||||
id: chart
|
||||
|
||||
@@ -71,9 +71,16 @@ jobs:
|
||||
fi
|
||||
}
|
||||
|
||||
# The three Go images build from the repo root and COPY
|
||||
# shared/ plus their own directory, so shared/ rebuilds all
|
||||
# three. proto/ is in server's list as insurance: the
|
||||
# The three Go images here build from the repo root and
|
||||
# COPY shared/ plus their own directory, so shared/ rebuilds
|
||||
# all three. vantagectl also depends on shared/ but is NOT
|
||||
# built here: it is a released tool, so its image is built and
|
||||
# version-tagged by vantagectl-release.yml on a vantagectl/v*
|
||||
# tag. A shared/ change therefore reaches it at the next
|
||||
# release rather than on the next push to main, which is the
|
||||
# point — an operator restoring a database should be running a
|
||||
# version they can name, not whatever main built last night.
|
||||
# proto/ is in server's list as insurance: the
|
||||
# generated pb is committed under server/, but a proto change
|
||||
# that someone regenerates in the same push should not depend
|
||||
# on that ordering.
|
||||
|
||||
@@ -0,0 +1,111 @@
|
||||
name: vantagectl Release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "vantagectl/v*"
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-docker
|
||||
container: node:26
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
go-version: "1.26"
|
||||
cache: true
|
||||
cache-dependency-path: vantagectl/go.sum
|
||||
|
||||
- name: Extract version
|
||||
id: version
|
||||
run: echo "VERSION=${GITHUB_REF_NAME#vantagectl/}" >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Test
|
||||
working-directory: vantagectl
|
||||
run: go test ./...
|
||||
|
||||
- name: Build
|
||||
working-directory: vantagectl
|
||||
env:
|
||||
VERSION: ${{ steps.version.outputs.VERSION }}
|
||||
run: |
|
||||
mkdir -p dist
|
||||
for target in linux/amd64 linux/arm64 darwin/arm64 windows/amd64; do
|
||||
goos="${target%/*}"
|
||||
goarch="${target#*/}"
|
||||
out="dist/vantagectl-${goos}-${goarch}"
|
||||
if [ "$goos" = "windows" ]; then out="${out}.exe"; fi
|
||||
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" go build \
|
||||
-ldflags="-s -w -X main.Version=${VERSION}" \
|
||||
-o "$out" .
|
||||
done
|
||||
|
||||
- name: Checksums
|
||||
working-directory: vantagectl/dist
|
||||
run: sha256sum vantagectl-* > checksums.txt
|
||||
|
||||
- name: Create release
|
||||
uses: https://gitea.com/actions/gitea-release-action@v1
|
||||
with:
|
||||
token: ${{ secrets.RELEASE_TOKEN }}
|
||||
files: |
|
||||
vantagectl/dist/vantagectl-linux-amd64
|
||||
vantagectl/dist/vantagectl-linux-arm64
|
||||
vantagectl/dist/vantagectl-darwin-arm64
|
||||
vantagectl/dist/vantagectl-windows-amd64.exe
|
||||
vantagectl/dist/checksums.txt
|
||||
|
||||
# The image is built here rather than in server-deploy.yml on every push to
|
||||
# main, because vantagectl is a released tool rather than a running service.
|
||||
# An operator restoring a database should be able to name the version they
|
||||
# ran; ":latest, rebuilt whenever main moved" cannot be named after the
|
||||
# fact. It is a separate job from the binaries because it needs a
|
||||
# docker-capable runner rather than a Go one, and it does not need the
|
||||
# binaries — the image builds from source in its own stage.
|
||||
image:
|
||||
runs-on: ubuntu-docker
|
||||
container: docker:dind
|
||||
steps:
|
||||
- name: Setup
|
||||
run: apk add --update nodejs npm git
|
||||
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Extract version
|
||||
id: version
|
||||
run: |
|
||||
# v0.1.0 for the binary stamp, 0.1.0 for the image tag: a
|
||||
# leading v is conventional on a git tag and unconventional on
|
||||
# a container tag.
|
||||
VERSION="${GITHUB_REF_NAME#vantagectl/}"
|
||||
echo "VERSION=${VERSION}" >> $GITHUB_OUTPUT
|
||||
echo "IMAGE_TAG=${VERSION#v}" >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Log in to registry
|
||||
run: |
|
||||
echo "${{ secrets.RELEASE_TOKEN }}" | \
|
||||
docker login ${{ vars.DOCKER_HOST }} \
|
||||
-u "${{ secrets.REGISTRY_USER }}" --password-stdin
|
||||
|
||||
- name: Build and push image
|
||||
env:
|
||||
VERSION: ${{ steps.version.outputs.VERSION }}
|
||||
IMAGE_TAG: ${{ steps.version.outputs.IMAGE_TAG }}
|
||||
run: |
|
||||
REPO="${{ vars.DOCKER_HOST }}/${{ github.repository_owner }}/vantage/vantagectl"
|
||||
# Root context: vantagectl depends on the shared module through
|
||||
# a replace directive, so the build needs shared/ alongside it.
|
||||
# VERSION is passed through so `vantagectl --version` inside the
|
||||
# image reports the tag it was built from rather than "dev".
|
||||
docker build \
|
||||
--build-arg VERSION="${VERSION}" \
|
||||
-t "${REPO}:${IMAGE_TAG}" \
|
||||
-t "${REPO}:latest" \
|
||||
-f vantagectl/Dockerfile .
|
||||
docker push "${REPO}:${IMAGE_TAG}"
|
||||
docker push "${REPO}:latest"
|
||||
@@ -91,7 +91,7 @@ vantage/
|
||||
│ └── models/ # accounts, instances, licences, plans
|
||||
├── adminsite/ # staff + customer console (vantage-hq)
|
||||
│ ├── app/(customer)/ # overview, instance, link, billing
|
||||
│ ├── app/(staff)/staff/ # operations, accounts, licences, plans, audit
|
||||
│ ├── app/(staff)/staff/ # operations, accounts, licences, pricing, audit
|
||||
│ ├── components/ # AppBar, PageHeader, PageFrame, InstanceRecord
|
||||
│ └── lib/ # api client, session guards, formatters
|
||||
├── docsite/ # user documentation (Docusaurus, static)
|
||||
@@ -315,6 +315,17 @@ a Service cannot address the one pod holding a console listener.
|
||||
|
||||
Agents report CPU/memory/swap/partitions/kernel — metrics every 30s, full static snapshot every 15 min. They also check for pending OS package updates hourly and can apply them on command (`ApplyUpdatesCmd`).
|
||||
|
||||
Windows update checking and applying go through the Windows Update COM API
|
||||
(`Microsoft.Update.Session`) rather than the PSWindowsUpdate module, which would
|
||||
need a PowerShell Gallery install on every host and fails on an air-gapped
|
||||
fleet. `CurrentVersion` is empty on Windows and `NewVersion` carries the KB
|
||||
article ID: a Windows update is not a version bump of a named package.
|
||||
|
||||
**The agent never reboots a host.** `ApplyUpdatesCmd` installs and stops there;
|
||||
`inventory.reboot_required` reports that one is owed, set on the static snapshot
|
||||
every 15 minutes. Linux fills it too, from `/var/run/reboot-required` or
|
||||
`dnf needs-restarting -r`.
|
||||
|
||||
### Package inventory and CVE findings
|
||||
|
||||
Agents report their installed packages hourly; the control plane matches them
|
||||
@@ -364,10 +375,24 @@ scheduler off entirely.
|
||||
|
||||
A **workload** is one Docker container or one systemd unit — one word for the
|
||||
page, the collection and the commands, rather than saying "container or
|
||||
service" in every identifier. Linux only, and **not gated by licence**: this
|
||||
reads as core fleet management, so v1 ships everywhere with no `HasFeature`
|
||||
check. If that changes the check belongs at `ReportWorkloads`, gating collection
|
||||
rather than display, exactly as sub-project A does.
|
||||
service" in every identifier.
|
||||
|
||||
On Windows a workload is a Docker container or a Windows **service**, reported
|
||||
under the same `unit` kind and the same `systemd_ok` / `systemd_error` fields —
|
||||
one wire shape, worded per platform in the UI, which is the only layer that
|
||||
knows the host's OS. The platform split lives entirely in the agent, as build
|
||||
tags (`systemd_linux.go` / `services_windows.go` and the matching `control_`
|
||||
and `logs_` pairs); the control plane is OS-blind and needed no changes.
|
||||
Windows collection runs PowerShell through `agent/internal/winexec`. Every
|
||||
script that reports data emits JSON that a build-tag-free parser reads, so
|
||||
those parsers are tested on Linux — the agent module has no Windows CI. The
|
||||
control verbs and `serviceDisplayName` emit no JSON and have no parser; they
|
||||
are exercised only by running the agent on Windows.
|
||||
|
||||
**Not gated by licence**: this reads as core fleet management, so v1 ships
|
||||
everywhere with no `HasFeature` check. If that changes the check belongs at
|
||||
`ReportWorkloads`, gating collection rather than display, exactly as sub-project
|
||||
A does.
|
||||
|
||||
Agents collect on a 60-second ticker and report through `ReportWorkloads` with
|
||||
the **offer-then-send** handshake the package report already uses. The offer is
|
||||
@@ -391,8 +416,9 @@ reads get `WorkloadLogsResult`. `CommandStream` republishes **every**
|
||||
no-op, so this costs nothing and avoids a second result path.
|
||||
|
||||
**The protected set is computed agent-side and enforced agent-side.**
|
||||
`vantage-agent.service`, plus the container ID read from `/proc/self/cgroup`
|
||||
should the agent ever run in a container. As with the console relay hardcoding
|
||||
`vantage-agent.service` on Linux, `VantageAgent` on Windows, plus the container
|
||||
ID read from `/proc/self/cgroup` should the agent ever run in a container. As
|
||||
with the console relay hardcoding
|
||||
`127.0.0.1`, the control plane may name a target but the agent decides what it
|
||||
will do to itself; a server-side denylist alone would be bypassed by the next
|
||||
dispatch path someone adds, and the failure is unrecoverable from the UI. The
|
||||
@@ -430,10 +456,213 @@ there are two copies — `agent/internal/grpc/pb` and `server/internal/grpc/pb`.
|
||||
A message added to one must be added to the other and to the `.proto`, in the
|
||||
same commit.
|
||||
|
||||
### Status pages
|
||||
|
||||
Two collections: `status_pages` is the page itself — title, banner, published
|
||||
flag, and an ordered list of sections each holding entries that pair a
|
||||
`monitor_id` with a per-page display name. `status_incidents` holds both
|
||||
operator-authored incidents and maintenance windows, sharing one document
|
||||
shape because they share a timeline, an impact and a set of affected
|
||||
components; each carries an explicit `page_ids` rather than deriving it from
|
||||
`affected_monitors`, because adding a monitor to a page later must not
|
||||
retroactively republish that monitor's old incidents to a new audience.
|
||||
|
||||
**`services.assembleSnapshot` is the redaction boundary, and it is the only
|
||||
one.** It takes a `snapshotInput` built from already-fetched
|
||||
`models.Monitor`/`models.Rollup`/`models.Incident` documents and returns a
|
||||
`StatusSnapshot` built entirely from a parallel, deliberately smaller
|
||||
vocabulary (`PublicComponent`, `PublicIncident`, …) that has no field for a
|
||||
target URL, host, port, expected status, keyword, failure message,
|
||||
certificate expiry, latency, runner or notification channel — `models.Monitor`
|
||||
itself never reaches an anonymous caller, only the handful of fields
|
||||
`assembleSnapshot` chooses to copy out of it. Being a pure function of already-
|
||||
fetched data (no DB calls inside it) is what makes the boundary testable
|
||||
without a database, which is the only thing standing between an editor adding
|
||||
a field to `PublicComponent` and that field being a hostname.
|
||||
|
||||
**An incident may only name components the page already carries.**
|
||||
`services.checkAffectedOnPages` refuses an `affected_monitors` entry that no
|
||||
page in the incident's `page_ids` lists, and the editor offers only the saved
|
||||
page's components — labelled by their per-page display name, since that is the
|
||||
name the reader sees. Naming an arbitrary monitor would publish a machine the
|
||||
page deliberately does not, which is the same leak `assembleSnapshot`'s
|
||||
redaction boundary exists to prevent, reached from the authoring side instead
|
||||
of the read side. It is a separate pass rather than part of `validateIncident`
|
||||
because it reads the database and `validateIncident` is a pure function of the
|
||||
document. A component dropped from the page **after** an incident named it
|
||||
makes the next edit of that incident fail, deliberately: the editor renders the
|
||||
stale entry flagged and checked so it is one click from being dropped, and the
|
||||
alternative is a page quietly publishing a component it no longer has.
|
||||
|
||||
Monitor-detected outages are **derived at read time, never copied**: each
|
||||
snapshot assembly reads recent `incidents` for the page's monitors and folds
|
||||
them into the timeline alongside the authored ones. There is no second
|
||||
incidents table for automatic ones and no reconciliation between two records
|
||||
of the same outage. A maintenance window in progress **repaints how a day is
|
||||
drawn, never the uptime number** — `buildDays` computes each day's up/down
|
||||
state and the 90-day percentage from rollups first, and
|
||||
`applyMaintenanceRepaint` only overwrites today's display state afterward, so
|
||||
a component that stayed up throughout a maintenance window still shows as up
|
||||
in its history.
|
||||
|
||||
The public route, `GET /public/status/:pageId`, is mounted on the gin **root**,
|
||||
outside `/api`, on purpose: `/api` carries `auth.Middleware`, `RequireScopes`,
|
||||
`RateLimitTokens` and `RequireActiveLicense` by virtue of where it is mounted,
|
||||
and a public route living there would need four exemptions — each one a hole a
|
||||
later change to any of those four could widen back open. A missing page, an
|
||||
unpublished page, and a page on the wrong host all answer the same 404;
|
||||
inventing a distinct code for "exists but unpublished" would itself leak that
|
||||
the page exists. A lapsed licence or a tier lacking `status_pages` answers 200
|
||||
with `available:false` and a `reason`, never a 403 or a blank page — the
|
||||
reader is a member of the public who can do nothing about either condition and
|
||||
deserves an explanation, not a browser error.
|
||||
|
||||
**The instance is resolved from `X-Forwarded-Host`, not `Host`.** The public
|
||||
page is server-rendered by `web/`, and the SSR fetch cannot set `Host` at all:
|
||||
it is a forbidden header name and undici drops it silently, so the Go server
|
||||
saw `server:8080` and every status page 404'd on every deployment. `web/`
|
||||
forwards the visitor's host in `X-Forwarded-Host` (and their address in
|
||||
`X-Forwarded-For`, or the whole deployment shares one rate-limit bucket), and
|
||||
`publicStatusInstance` honours that header **only when `c.RemoteIP()` is in
|
||||
`TRUSTED_PROXIES`** — it selects a tenant, so an untrusted peer must not be
|
||||
able to name one. It uses `RemoteIP()` and not `ClientIP()` deliberately: the
|
||||
latter is reconstructed from the very headers being judged.
|
||||
|
||||
**A host naming no slug falls back to the sole instance on a non-cloud
|
||||
deployment.** `hostSlug` requires `<slug>.vantage.<tld>`; a self-hosted install
|
||||
at `vantage.acme.com` or an IP has no slug and would otherwise 404 forever. It
|
||||
has exactly one instance, resolved with the same count-then-read bootstrap
|
||||
uses, cached alongside the slug lookups. More than one instance is a 404, not a
|
||||
guess. A host that *does* name a slug which does not exist stays a 404 —
|
||||
falling back there would serve one tenant's page on another's address.
|
||||
|
||||
Assembled snapshots are cached in Redis for **30 seconds**, keyed per
|
||||
instance and page, and every authoring write (`UpdateStatusPage`,
|
||||
`DeleteStatusPage`, and every incident mutation) invalidates its page's entry
|
||||
immediately rather than waiting out the TTL — an operator posting an update
|
||||
mid-incident should not wonder for half a minute whether it saved. A cache
|
||||
miss, on Redis being down or on any read error, degrades to reassembly rather
|
||||
than an error: the status page has to survive the outage it exists to report.
|
||||
The public endpoint itself is rate limited to **120 requests per minute per
|
||||
client address**, answering 429 with `Retry-After`, on the same fixed-window
|
||||
pattern as `RateLimitTokens`.
|
||||
|
||||
**`TRUSTED_PROXIES` is load-bearing for that limiter, not cosmetic.** `main.go`
|
||||
always calls `gin.SetTrustedProxies` with it; left unset, gin trusts no proxy
|
||||
and `c.ClientIP()` falls back to the direct peer address — which, sat behind a
|
||||
real reverse proxy, is the proxy's own address for every visitor. The rate
|
||||
limiter then keys on one address for the whole fleet of readers, and the first
|
||||
burst of legitimate traffic during an incident is what trips it. Set it to the
|
||||
proxy's real address or CIDR, not merely a private range guess; the shipped
|
||||
compose file and Helm chart default it to the RFC1918 ranges, which is right
|
||||
for their own bundled reverse proxy but wrong the moment another one is
|
||||
inserted in front. The same setting also decides the address recorded in
|
||||
`audit_logs` and `console_sessions`.
|
||||
|
||||
### Agent self-update
|
||||
|
||||
`UpdateAgentCmd` carries a target version and Gitea base URL; the agent downloads and replaces itself.
|
||||
|
||||
### Backup and restore
|
||||
|
||||
`vantagectl` is a standalone Go module (`vantagectl/`), not a subcommand of
|
||||
`server`. It needs its own module rather than living inside `server`'s for the
|
||||
same reason `admin` and `sitesvc` already do: `server` imports the rest of
|
||||
`server`'s dependency graph, and `spf13/cobra` has no business in a process
|
||||
that also terminates gRPC streams and serves the REST API. More to the point,
|
||||
`vantagectl` has to run when the control plane **does not** — a backup or
|
||||
restore against a database with no server container alive at all — so it
|
||||
cannot be a mode of the binary whose crash is the reason you need it.
|
||||
|
||||
The actual logic lives in `shared/backup` (dump, restore, verify, manifest,
|
||||
fingerprint), not in `vantagectl/internal/cmd`, which holds only argument
|
||||
parsing and operator-facing output. That split is what lets `server` import
|
||||
`shared/backup` later — a scheduled in-process backup, say — without a second
|
||||
implementation to keep in sync. `shared/cryptobox` is the same move one layer
|
||||
down: it is now the **single** AES-256-GCM implementation, and
|
||||
`server/internal/services/crypto.go` delegates to it rather than keeping its
|
||||
own copy that `shared/backup` would otherwise have had to duplicate to decrypt
|
||||
a probe value during `verify`.
|
||||
|
||||
**The archive stores a SHA-256 fingerprint of `KEY_ENCRYPTION_KEY`, never the
|
||||
key.** `backup` refuses to run without the key set in the environment unless
|
||||
`--allow-no-key` is passed, because an archive with no fingerprint at all
|
||||
cannot later tell a restore that the wrong key is in hand — it can only find
|
||||
that out when the data comes back as noise. The fingerprint is what turns that
|
||||
failure into a refusal at `restore` time instead.
|
||||
|
||||
**Collections are enumerated live** — `shared/backup` lists what the database
|
||||
actually holds rather than reading `services.ScopedCollections`, the opposite
|
||||
choice from the one instance-deletion purge makes. Purge must never miss a
|
||||
tenant-scoped collection, so it keeps one hand-maintained registry; a backup
|
||||
must never miss **any** collection, tenant-scoped or not (`migrations`,
|
||||
`vulndb_meta`), so a static list is the wrong shape twice over — once for the
|
||||
collections it would still owe `instance_id` deletion but not a backup, and
|
||||
once for the two singleton collections that carry neither `instance_id` nor a
|
||||
release note.
|
||||
|
||||
**Restore refuses a non-empty target database and has no merge semantics.**
|
||||
There is no code path that upserts an archive's documents over existing ones:
|
||||
merging two control planes' data reconciles nothing about which SSH keys are
|
||||
still valid or which users still exist, and an upsert would resurrect a
|
||||
revoked key or a deleted member from the older side. `--force` drops each
|
||||
collection in the archive first, and is gated behind a second assurance:
|
||||
`--confirm-db NAME` matching the target exactly, which works everywhere, or —
|
||||
on a terminal only, and only when `--confirm-db` was not given — the target
|
||||
database's name typed back at a prompt. `--confirm-db` is accepted on a
|
||||
terminal too: it is the stronger of the two, because naming the target in the
|
||||
command itself means a copied command carries its intended target with it and
|
||||
cannot destroy a different one by accident. Without a terminal and without
|
||||
`--confirm-db`, `--force` is refused.
|
||||
|
||||
**`--force` drops only what the archive names.** Collections already in the
|
||||
target that the archive does not carry are left untouched and **named in a
|
||||
warning** — an archive taken with `--exclude workflow_log_lines` restored over
|
||||
a live database leaves the old lines joined to restored runs, which the
|
||||
operator must be told. Dropping them instead would delete data nobody asked to
|
||||
delete, and there is no way back from that.
|
||||
|
||||
**Index specifications are replayed verbatim, never reconstructed.**
|
||||
`dumpIndexes` stores each spec as extended JSON over the raw BSON the server
|
||||
reported, and `replayIndexes` hands it back to `createIndexes` through
|
||||
`RunCommand` with only `v` and `ns` stripped and `_id_` skipped. Rebuilding a
|
||||
`mongo.IndexModel` from a hand-picked set of options dropped
|
||||
`partialFilterExpression` — which this codebase relies on in
|
||||
`services/workflows.go` and `services/settings.go` — so a partial unique index
|
||||
came back as a full one, failed on duplicate keys, and aborted the restore
|
||||
mid-write. Reconstructing the key document from JSON also lost compound key
|
||||
order, which is significant.
|
||||
|
||||
**`backup.ciphertextFields` mirrors `server/internal/models` by hand.**
|
||||
`shared/` is a separate module and `models` is under `server/internal`, so
|
||||
`shared/backup` cannot import it; the map naming each collection's `*_enc`
|
||||
fields (`keys`, `secrets`, `auth_providers`, `console_sessions`) must change in
|
||||
the same commit as any of those bson tags, the same hazard as
|
||||
`web/lib/targets.ts` and `services.MaxWorkloadLogLines`. Wrong field names are
|
||||
silent: `verify`'s live probe simply finds no ciphertext and reports "this
|
||||
database stores no ciphertext yet", so the one gate that catches what a
|
||||
fingerprint cannot no-ops. `settings` is deliberately in neither that map nor
|
||||
`CiphertextCollections()` — its ESO read token is a SHA-256 hash, not
|
||||
ciphertext.
|
||||
|
||||
**A file-backed `backup` writes to `<name>.tar.gz.partial` and renames on
|
||||
success**, the same discipline the agent uses for `authorized_keys`. A failed
|
||||
dump must not leave a partial file named exactly like a good archive; `--out -`
|
||||
is untouched, since a broken pipe has no file to mislead anyone.
|
||||
|
||||
**`vantagectl/Dockerfile`'s runtime stage is `scratch`, and needs the same
|
||||
explicit `/tmp` as `server/Dockerfile`.** `restore` extracts an archive to a
|
||||
temporary directory before verifying its checksums, and a scratch image has no
|
||||
`/tmp` for `os.MkdirTemp` to find — the same failure mode `vulnsched` hits on
|
||||
`server`, but here it would break every restore rather than only vulnerability
|
||||
scanning.
|
||||
|
||||
**`shared/` reaches four Go images, but only three of them from
|
||||
`server-deploy.yml`** (`server`, `sitesvc`, `admin`). The `vantagectl` image is
|
||||
built by `vantagectl-release.yml` on a `vantagectl/v*` tag instead, so a
|
||||
`shared/` change reaches it at the next release rather than the next push to
|
||||
main — see the CI section below.
|
||||
|
||||
### API tokens and OpenAPI
|
||||
|
||||
A token is `vt_` plus 32 random bytes hex, shown once at creation and stored
|
||||
@@ -441,9 +670,9 @@ only as sha256 — the same shape as `servers.agent_token_hash` and the ESO read
|
||||
token, and for the same reason: nothing downstream ever needs the plaintext
|
||||
back. It belongs to the user who created it, and its role can never exceed
|
||||
theirs; see the `api_tokens` note under MongoDB Collections for how that stays
|
||||
true across a demotion rather than only at issuance. Scopes are eight
|
||||
true across a demotion rather than only at issuance. Scopes are nine
|
||||
resources — `servers`, `keys`, `secrets`, `workflows`, `monitors`, `vulns`,
|
||||
`workloads`, `settings` — each split into `:read` and `:write`, with `:write`
|
||||
`workloads`, `status`, `settings` — each split into `:read` and `:write`, with `:write`
|
||||
satisfying a `:read` requirement on the same resource so a caller does not have
|
||||
to hold both. Any signed-in member may mint and revoke their **own** tokens —
|
||||
there is no `RequireRole` on `POST /tokens` or `DELETE /tokens/:id` — because
|
||||
@@ -722,6 +951,10 @@ workloads GET /workloads · GET /servers/:id/workloads
|
||||
POST /servers/:id/workloads/refresh
|
||||
POST /servers/:id/workloads/:wid/action (owner|admin)
|
||||
GET /servers/:id/workloads/:wid/logs (owner|admin)
|
||||
status-pages GET,POST /status-pages · GET,PUT,DELETE /status-pages/:pageId (owner|admin)
|
||||
GET,POST /status-pages/:pageId/incidents
|
||||
PUT,DELETE /status-pages/:pageId/incidents/:incidentId
|
||||
POST /status-pages/:pageId/incidents/:incidentId/updates
|
||||
audit GET /audit
|
||||
agent GET /agent/latest-version
|
||||
settings GET,PUT /settings · POST /settings/secrets-token (owner|admin)
|
||||
@@ -805,7 +1038,7 @@ Paddle is merchant of record; `admin/internal/paddle` is a thin REST client (no
|
||||
|
||||
## MongoDB Collections
|
||||
|
||||
`servers` · `keys` · `assignments` · `orgs` · `users` · `auth_providers` · `settings` · `secrets` · `workflows` · `workflow_steps` · `workflow_runs` · `workflow_log_lines` · `workflow_log_seq` · `monitors` · `incidents` · `monitor_rollups` · `notification_channels` · `console_sessions` · `audit_logs` · `server_packages` · `vuln_findings` · `vuln_alert_rules` · `vulndb_meta` · `server_workloads` · `api_tokens` · `migrations`
|
||||
`servers` · `keys` · `assignments` · `orgs` · `users` · `auth_providers` · `settings` · `secrets` · `workflows` · `workflow_steps` · `workflow_runs` · `workflow_log_lines` · `workflow_log_seq` · `monitors` · `incidents` · `monitor_rollups` · `notification_channels` · `console_sessions` · `audit_logs` · `server_packages` · `vuln_findings` · `vuln_alert_rules` · `vulndb_meta` · `server_workloads` · `api_tokens` · `status_pages` · `status_incidents` · `migrations`
|
||||
|
||||
Every document except `migrations` carries `org_id`. Struct definitions are the source of truth — see `server/internal/models/`.
|
||||
|
||||
@@ -828,7 +1061,9 @@ Notes that are not obvious from the structs:
|
||||
|
||||
Admin's own database is separate and holds `accounts` · `admin_instances` · `licenses` · `subscriptions` · `plans` · `catalogue` · `entitlements` · `paddle_events` · `staff_users` · `customer_users` · `instance_members` · `admin_audit`. `paddle_events` is the webhook idempotency log, unique on `event_id`: an event is claimed there before processing, and a duplicate of a handled event is a 200 no-op. `instance_members` is unique on `(instance_id, customer_user_id)` — one person holds at most one user in one instance, which makes a grant idempotent-by-refusal rather than silently doubling a projection. It is an _index_ of the control-plane rows, not the authority (see "Grants project, they do not federate"). Admin has no migrations collection; `models.Backfill` runs on every boot and is idempotent by filtering on the absence of what it writes.
|
||||
|
||||
`plans` is keyed on `(deployment, tier)` — six rows, two deployments times three tiers — and holds base allowances only. **Every Paddle price ID lives in `catalogue`**, one row per priceable component (`base`, `limit`, `feature`), because a metered plan is priced by several prices and one map on a plan row cannot express that. `entitlements` holds one row per instance with `desired` beside `granted`: the checkout is built from `desired`, a licence is only ever signed from `granted`, and an abandoned checkout therefore leaves a `desired` that reached nothing. The two Free plans have **no catalogue rows at all**, which is what keeps Free outside Paddle.
|
||||
`plans` is keyed on `(deployment, tier)` — six rows, two deployments times three tiers — and holds base allowances only. **Every Paddle price ID lives in `catalogue`**, one row per priceable component (`base`, `limit`, `feature`), because a metered plan is priced by several prices and one map on a plan row cannot express that. A row carries a `scope`: `plan` rows name a `deployment` and `tier` and belong to that plan alone, `shared` rows leave both empty and are sold by every paid plan. **How many rows a component needs follows from how many Paddle products it is** — the base fee is a different product per plan, every add-on is one product at one price, so the catalogue is four base rows plus five shared rows, nine instead of twenty-four, and an add-on's price ID is typed once rather than four times. `models.CatalogueFor` is the seam: it returns a plan's base row plus every shared row, and **nothing may filter the catalogue by `deployment` and `tier` itself** or it sees a plan priced by its base fee alone. `adminsite/lib/catalogue.ts`'s `rowsForPlan` is the TypeScript half of that and must change in the same commit, the same shape of hazard as `web/lib/targets.ts`. `models.MigrateSharedCatalogue` runs at boot after `SeedCatalogue`, merges the old per-plan copies onto the shared row and deletes them; it **refuses rather than guesses** when the four copies disagree, because four rows meant to be one price and are not is a pricing decision somebody made and picking one silently moves a customer's bill. `entitlements` holds one row per instance with `desired` beside `granted`: the checkout is built from `desired`, a licence is only ever signed from `granted`, and an abandoned checkout therefore leaves a `desired` that reached nothing. The two Free plans have **no catalogue rows at all**, which is what keeps Free outside Paddle.
|
||||
|
||||
**No tier bundles a feature.** `console`, `oidc`, `vuln_scanning` and `status_pages` are each a per-customer priceable add-on: every plan row carries an empty `base_features`, and the grant comes from a `catalogue` row the customer buys. Adding a fifth feature therefore means one more shared `KindFeature` row in `SeedCatalogue`'s `seedRows` and one entry in `adminsite/lib/features.ts` — that map is what the customer's grant list, the staff configurator and the purchase form all enumerate, so a feature missing from it exists in the licence and is invisible in the portal. `SeedCatalogue` upserts on the row's natural key `(kind, deployment, tier, limit_key, feature_key)` — a shared row's empty deployment and tier are part of that key, not a wildcard — so a new row reaches an existing database on the next admin boot with no migration; `SeedPlans` is `$setOnInsert` on the whole document and would not, which is the other reason bundling into a tier is the harder path.
|
||||
|
||||
### Migrations
|
||||
|
||||
@@ -871,7 +1106,7 @@ tls: true
|
||||
|
||||
```
|
||||
1. SyncKeys(server_id, agent_token, agent_version)
|
||||
2. Non-Linux hosts stop here — Windows agents register and heartbeat only
|
||||
2. Non-Linux hosts stop here — the key-management steps below are Linux-only; a Windows agent's other work (workflow steps, inventory, OS updates, workloads) runs from the goroutines started above, not from this loop
|
||||
3. Diff desired keys against /root/.ssh/authorized_keys; unchanged → no write
|
||||
4. Changed → write .tmp, os.Rename() over the real file, chmod 0600
|
||||
```
|
||||
@@ -914,6 +1149,7 @@ Windows: MSI built by CI (WiX), or `installer/setup.ps1` registering the agent a
|
||||
| `PROXY_ADVERTISE_HOST` | no | default `server`; the hostname guacd resolves the control plane by, handed to guacd as the relay's address. Wrong here and every console session fails at connect |
|
||||
| `PROXY_LISTEN_HOST` | no | default `0.0.0.0`; the interface the ephemeral relay listener binds |
|
||||
| `APP_ROOT_LABEL` | no | default `vantage`; wrong value disables the host/session org guard |
|
||||
| `TRUSTED_PROXIES` | no | comma-separated CIDRs or addresses gin trusts for `X-Forwarded-For`. Empty means trust none: `c.ClientIP()` falls back to the direct peer address, which behind a real reverse proxy is that proxy's own address for every visitor — the public status page's per-address rate limit then keys on one address for the whole fleet of readers. Also the address recorded in `audit_logs` and `console_sessions`. Compose and the Helm chart default it to the RFC1918 ranges, right for their own bundled proxy and wrong the moment another one is inserted in front |
|
||||
| `POD_IP` | no | this pod's own address, set by the Helm chart from the downward API. **Takes precedence over `PROXY_ADVERTISE_HOST`** — a console relay listener belongs to one replica, and a Service address names all of them |
|
||||
| `VANTAGE_MIGRATE_ONLY` | no | run schema setup (migrations, index builders, default-step seeding) and exit without serving. `GRPC_HOST` is not required in this mode. Set by the Helm chart's pre-upgrade Job |
|
||||
| `VANTAGE_SKIP_MIGRATIONS` | no | serve without running schema setup, on the assumption a Job already did. Set by the chart's Deployment whenever `server.migrationJob.enabled`. Unset under Compose, where one process still migrates and then serves |
|
||||
@@ -945,7 +1181,7 @@ Windows: MSI built by CI (WiX), or `installer/setup.ps1` registering the agent a
|
||||
|
||||
**`ingress.web.host` is normally a wildcard.** `*.vantage.example.com` is the per-tenant instance namespace — `APP_ROOT_LABEL` resolves the instance from the label. A Kubernetes wildcard host matches **exactly one** label, so it does not match the apex, and here that is correct rather than a gap: `vantage.hostxtra.co.uk` is the marketing site (`site/`, in `docker-compose.site.yml`), which this chart does not deploy. `extraHosts` is for a genuine second name; adding the apex to it would put the control plane on the marketing host. Every host in the list gets identical paths.
|
||||
|
||||
**`ingress.api.enabled` routes `/api` and `/auth` straight to the server.** Both arrangements work — without it `web` proxies those prefixes onward itself (`web/next.config.ts`) — but edge routing is one hop shorter and matches what the Nginx Proxy Manager in front of the Docker deployment already does, so leaving it off makes the request path a different shape on Kubernetes than in production. It stays **off by default** because it only helps where the server is reachable on the same host and certificate as `web`; turning it on blindly moves the whole API onto a route that may not be provisioned. Traefik derives router priority from rule length, so `PathPrefix(/api)` outranks the catch-all `/` with no priority annotation needed.
|
||||
**`ingress.api.enabled` routes `/api`, `/auth`, `/public`, `/install*` and `/update*` straight to the server, and it is not optional.** It defaults to **true** and the chart refuses to render with it off, because `web` proxies nothing: with those prefixes unrouted the UI loads and every request it makes 404s against Next. The value survives only for an installation whose own terminator sits in front of this ingress and routes them there instead. Traefik derives router priority from rule length, so `PathPrefix(/api)` outranks the catch-all `/` with no priority annotation needed.
|
||||
|
||||
**The gRPC route needs its own Service.** The server terminates no TLS; it speaks plain h2c and always has, with TLS terminated by whatever sits in front. Traefik will not use h2c to a backend unless the *Service* says so, and that annotation applies to every port on the Service — so annotating the shared two-port `<release>-server` would force h2c on its HTTP port too.
|
||||
|
||||
@@ -955,6 +1191,8 @@ TLS is `ingress.tls.secretName` / `grpcSecretName` (pre-existing certificates) *
|
||||
|
||||
---
|
||||
|
||||
**Neither compose file ships a reverse proxy, and both now need one.** `web:3000` serves the UI only; a request to `/api` there is a Next 404. Route `/api`, `/auth`, `/public`, `/install`, `/install.ps1`, `/update`, `/update.ps1` to `server:8080` and everything else to `web:3000` — on vantage.hostxtra.co.uk that is the Nginx Proxy Manager already in front, and it is what a self-hosted install has to configure before the UI works at all.
|
||||
|
||||
`deploy/docker-compose.yml` runs four services: `redis`, `guacd`, `server` (8080 + 9090), `web` (3000). MongoDB is external. `deploy/docker-compose.site.yml` adds five more — `site` (3003), `sitesvc` (8082), `admin` (8083), `adminsite` (3004) and `docsite` (3005) — and is only used on vantage.hostxtra.co.uk.
|
||||
|
||||
`docsite` is the odd one: a **static** build served by `nginx:alpine-slim`, not a Node runtime, and it listens on `80` rather than `3000`. It is reached at **`vantage.hostxtra.co.uk/docs`** — a path on the marketing host, routed by its own Nginx Proxy Manager location, which must sort **above** the catch-all forwarding to `site:3003` or Next answers the 404. A path and not a subdomain because `*.vantage.hostxtra.co.uk` is the per-tenant instance namespace and `APP_ROOT_LABEL` would read a `docs.` label as a tenant slug. NPM forwards the **full** path upstream — it does not strip `/docs` — so `DOCS_BASE_URL`, the proxy location and the directory the image copies the build into (`/usr/share/nginx/html/docs`) must all agree. When they do not, the HTML loads and every asset 404s.
|
||||
@@ -1004,6 +1242,27 @@ Tailwind in all three maps `var(--…)` references only, so **no component in an
|
||||
|
||||
`web/` collapses Tailwind's radius scale — `md`, `lg` and `xl` all resolve to site/'s 4px — rather than rewriting the ~140 `rounded-lg` classes across its pages. Every one of them meant "a panel corner", and `tailwind.config.ts` is now where that decision lives. `rounded-full` is untouched: status dots and pills still need it.
|
||||
|
||||
**Plans and the catalogue are one page, `/staff/pricing`.** They were two nav
|
||||
entries and the split asked staff to hold one half in their head while looking
|
||||
at the other: a tier's allowance is what the metered component charges above,
|
||||
and a base fee means nothing without the allowance it includes. The page is
|
||||
`PlansSection` then `CatalogueSection`, in the order the decision is made —
|
||||
what a tier grants, then what it costs. `next.config.ts` keeps permanent
|
||||
redirects from `/staff/plans` and `/staff/catalogue`, which are bookmarked in
|
||||
staff browsers. **The tier list is cards, not forms**: six plans with five
|
||||
number fields, a select, a checkbox and four feature toggles each was forty-odd
|
||||
controls on one screen, and the page could not be read for the thing it exists
|
||||
to answer. A card states what the tier grants and `Modal` — a native
|
||||
`<dialog>`, for the focus trap and Escape handling a hand-rolled overlay gets
|
||||
wrong — is where it is changed. Every feature key renders on every card, lit or
|
||||
unlit: no tier bundles one today, so the unlit row is the information.
|
||||
|
||||
**The catalogue's coverage ledger is not decoration.** A missing production
|
||||
price is invisible in a grid of text inputs — every cell looks like every other
|
||||
until twenty-six characters of each are read — and it is the one thing staff
|
||||
come to the page to check before a launch, so each component draws one filled
|
||||
or empty square per environment and term.
|
||||
|
||||
**The `adminsite/` shell.** `AppBar` is the single masthead — identity, nav, environment, account menu — and it belongs to the two authenticated layouts, never to `app/layout.tsx`, so `/login` and `/accept-invite` do not render navigation they cannot use. Nav active state is derived from `usePathname`; do not hardcode it. `PageHeader` gives every screen the same back link, title, actions and **record line** (the reference number in mono, click-to-copy) — the reference is what people paste into support tickets, so it has a fixed slot rather than a per-page treatment. `PageFrame` is the main-plus-320px-rail split; the rail carries only what is true account-wide, which is why there is no plan card in it — **tier, limits and expiry belong to a licence, and a licence belongs to one instance**, so an account holding a Free cloud instance and a Professional self-hosted one has no single plan.
|
||||
|
||||
Customer nav is three destinations — Overview, People, Billing. Settings is in the account menu because it is your password, not a place, and appearance lives there too: `AccountMenu` is the only thing that sets `data-theme`, which the token blocks have always supported in both directions.
|
||||
@@ -1062,7 +1321,7 @@ GOOS=linux GOARCH=amd64 go build \
|
||||
|
||||
### `server-deploy.yml` — triggered on every push to `main`
|
||||
|
||||
Builds and pushes seven images to the Gitea container registry: `server`, `web`, `site`, `sitesvc`, `admin`, `adminsite` and `docsite`.
|
||||
Builds and pushes seven images to the Gitea container registry: `server`, `web`, `site`, `sitesvc`, `admin`, `adminsite` and `docsite`. **`vantagectl` is deliberately not among them** — it is a released tool rather than a running service, and its image is version-tagged by `vantagectl-release.yml`.
|
||||
|
||||
Note that despite the name, **this workflow does not deploy** — it only builds and pushes. There is no SSH step. Rolling images out is a separate manual step on the host:
|
||||
|
||||
@@ -1080,7 +1339,17 @@ cd /opt/vantage && docker compose -f docker-compose.yml -f docker-compose.site.y
|
||||
| `sitesvc` | `sitesvc/`, `shared/`, `go.work` |
|
||||
| `web` · `site` · `adminsite` · `docsite` | their own directory only |
|
||||
|
||||
`shared/` fans out to all three Go images because each of their Dockerfiles copies `shared/` from a root context — **if a fourth service ever imports `shared/`, add it to that list or it will ship stale**. A change to the workflow file rebuilds everything, since a build arg is baked into the image. So does anything that leaves no trustworthy base commit: a manual `workflow_dispatch`, a new branch, or a force-push whose old head is gone.
|
||||
`shared/` fans out to **three** images here (`server`, `sitesvc`, `admin`)
|
||||
because each of their Dockerfiles copies `shared/` from a root context — **if a
|
||||
fourth service ever imports `shared/`, add it to that list or it will ship
|
||||
stale**. `vantagectl` also imports `shared/` and is the exception: it is built
|
||||
by `vantagectl-release.yml`, so a `shared/` fix reaches it only when someone
|
||||
cuts a `vantagectl/v*` tag. That is deliberate — an operator restoring a
|
||||
database should be running a version they can name — but it does mean a
|
||||
`shared/backup` fix is not live until it is released. A change to the workflow file rebuilds everything, since
|
||||
a build arg is baked into the image. So does anything that leaves no
|
||||
trustworthy base commit: a manual `workflow_dispatch`, a new branch, or a
|
||||
force-push whose old head is gone.
|
||||
|
||||
The gap this leaves: **changing a repo variable pushes no commit, so nothing rebuilds.** After editing `ADMIN_API_URL`, `HQ_URL` or `ADMIN_ENV`, run the workflow manually — that is what `workflow_dispatch` is there for. Base images also stop being refreshed on a service nobody touches; a periodic manual run covers that.
|
||||
|
||||
@@ -1113,7 +1382,7 @@ git push origin main # server + web deploy
|
||||
| `REGISTRY_USER` | Secret | Gitea username. Must own `RELEASE_TOKEN`, or basic auth is rejected |
|
||||
| ~~`REGISTRY_PASSWORD`~~ | — | **Not used.** Named here historically; no workflow reads it. Referencing an unset secret yields an empty password and a `401 Failed to authenticate user` that looks like a token scope problem. Use `RELEASE_TOKEN` |
|
||||
| `DOCKER_HOST` | Variable | registry host used for image tags |
|
||||
| `API_URL` | **not** a CI variable | `web` reads it at **runtime**, from the container environment — `next.config.ts` is evaluated when `server.js` boots in standalone mode, and the rewrites it feeds are server-side, never browser-side. Default `http://localhost:8080`; compose sets `http://server:8080`. `NEXT_PUBLIC_API_URL` is still honoured as a fallback for existing deployments. |
|
||||
| ~~`API_URL`~~ | — | **Gone.** `web` proxies nothing and holds no address for the control plane. `/api`, `/auth`, `/public`, `/install*` and `/update*` must be routed to `server:8080` by the reverse proxy in front of both; everything else goes to `web:3000`. One variable that could name the wrong host was one request path too many — pointed at the marketing site, `/public/status/…` answered a Next 404 indistinguishable from a status page that does not exist. |
|
||||
| `SITE_API_URL` | Variable | **browser-reachable** sitesvc URL, baked into the `site` image. Required — if empty, both forms report "not connected" and submit nowhere. Must also be in sitesvc's `SITE_ORIGIN`. |
|
||||
| `SITE_CONTACT_EMAIL` | Variable | optional; address shown when a form is misconfigured |
|
||||
| `SITE_URL` | Variable | browser URL of the marketing site, baked into `adminsite` so `/login` can point at `/start`. **Signup has no page in `adminsite` at all** — one signup form, on `site/`. Empty renders no link rather than one that 404s. |
|
||||
@@ -1144,7 +1413,14 @@ git push origin main # server + web deploy
|
||||
- **guacd for console** — protocol handling is Guacamole's problem, not ours; we proxy the WebSocket and manage credentials.
|
||||
- **`org_id` on every document** — isolation enforced at the query layer, not by separate databases.
|
||||
- **root only** — manages `/root/.ssh/authorized_keys`; no per-user key management.
|
||||
- **Windows agents are second-class by design** — register, heartbeat, run steps, report inventory; no `authorized_keys` management.
|
||||
- **Windows agents cover the fleet-management path** — register, heartbeat, run
|
||||
steps, report inventory, OS updates through the Windows Update COM API, and
|
||||
workloads (services plus containers, with control and logs). They still do no
|
||||
`authorized_keys` management, and no package inventory or CVE matching: a
|
||||
Windows agent never calls `ReportPackages`, so no `server_packages` document
|
||||
exists for it and it reports no package inventory at all — a different,
|
||||
earlier state than the `unsupported` a Linux distribution reaches when its
|
||||
family has no security feed.
|
||||
- **Both `server` and `web` scale horizontally** — see "Running more than one server replica" below. `web` holds nothing; `server` holds per-agent state that is routed between replicas over Redis rather than duplicated.
|
||||
- **Deletion lives in the control plane** — admin sends the warnings because it knows the billing address; the control plane performs the delete because it is the only service that knows which collections carry `instance_id`. Mirroring that list into admin would drift, and a drift there deletes the wrong rows.
|
||||
|
||||
|
||||
@@ -86,6 +86,10 @@ func main() {
|
||||
idxCancel()
|
||||
log.Fatalf("seed catalogue: %v", err)
|
||||
}
|
||||
if err := models.MigrateSharedCatalogue(idxCtx); err != nil {
|
||||
idxCancel()
|
||||
log.Fatalf("migrate catalogue: %v", err)
|
||||
}
|
||||
if err := models.Backfill(idxCtx); err != nil {
|
||||
idxCancel()
|
||||
log.Fatalf("backfill: %v", err)
|
||||
|
||||
@@ -485,6 +485,7 @@ func staffListCatalogue(c *gin.Context) {
|
||||
func staffUpdateCatalogue(c *gin.Context) {
|
||||
var body struct {
|
||||
Kind string `json:"kind"`
|
||||
Scope string `json:"scope"`
|
||||
Deployment string `json:"deployment"`
|
||||
Tier string `json:"tier"`
|
||||
LimitKey string `json:"limit_key"`
|
||||
@@ -500,11 +501,19 @@ func staffUpdateCatalogue(c *gin.Context) {
|
||||
// mean a resolved self-hosted monthly price later, which the resolver treats
|
||||
// as a configuration error — better to refuse it at the point somebody
|
||||
// pastes it, while they are looking at the screen.
|
||||
//
|
||||
// A shared row is sold by both deployments, so both terms are legitimate on
|
||||
// it: the cloud checkout takes the monthly price and the self-hosted one
|
||||
// never asks for it. Only a plan row can name a term its own deployment
|
||||
// does not sell.
|
||||
for env, byTerm := range body.PriceIDs {
|
||||
for term, id := range byTerm {
|
||||
if id == "" {
|
||||
continue
|
||||
}
|
||||
if body.Scope == models.ScopeShared {
|
||||
continue
|
||||
}
|
||||
if !termSold(body.Deployment, term) {
|
||||
c.JSON(http.StatusBadRequest, gin.H{
|
||||
"error": fmt.Sprintf("%s does not sell %s (environment %s)",
|
||||
@@ -514,6 +523,8 @@ func staffUpdateCatalogue(c *gin.Context) {
|
||||
}
|
||||
}
|
||||
|
||||
// Addressed by its natural key, so the staff UI never holds a Mongo id. A
|
||||
// shared row's empty deployment and tier are part of that key.
|
||||
filter := bson.M{
|
||||
"kind": body.Kind,
|
||||
"deployment": body.Deployment,
|
||||
@@ -534,12 +545,21 @@ func staffUpdateCatalogue(c *gin.Context) {
|
||||
audit.Write(c.Request.Context(), models.AuditEntry{
|
||||
Actor: auth.Current(c).Email,
|
||||
Action: "catalogue.updated",
|
||||
Target: body.Deployment + "/" + body.Tier + "/" + body.Kind,
|
||||
Target: catalogueTarget(body.Scope, body.Deployment, body.Tier, body.Kind),
|
||||
Detail: body.LimitKey + body.FeatureKey,
|
||||
})
|
||||
c.JSON(http.StatusOK, gin.H{"updated": true})
|
||||
}
|
||||
|
||||
// catalogueTarget names an edited component in the audit log. A shared row has
|
||||
// no plan to name, so it says so rather than logging "//feature".
|
||||
func catalogueTarget(scope, deployment, tier, kind string) string {
|
||||
if scope == models.ScopeShared {
|
||||
return "shared/" + kind
|
||||
}
|
||||
return deployment + "/" + tier + "/" + kind
|
||||
}
|
||||
|
||||
func termSold(deployment, term string) bool {
|
||||
for _, t := range license.TermsFor(deployment) {
|
||||
if t == term {
|
||||
|
||||
@@ -29,8 +29,18 @@ func LineItems(ctx context.Context, env, term string, plan *models.Plan, cfg mod
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(rows) == 0 {
|
||||
return nil, fmt.Errorf("%w: %s/%s is priced by nothing",
|
||||
// A plan is identified by its base row, and shared add-on rows exist whether
|
||||
// or not any plan sells them — so "the catalogue returned something" is no
|
||||
// longer proof this plan is priced. Check for the base row itself.
|
||||
hasBase := false
|
||||
for _, r := range rows {
|
||||
if r.Kind == models.KindBase {
|
||||
hasBase = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hasBase {
|
||||
return nil, fmt.Errorf("%w: %s/%s has no base row",
|
||||
ErrUnpriced, plan.Deployment, plan.Tier)
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,8 @@ package models
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/admin/internal/db"
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/shared/license"
|
||||
@@ -19,6 +21,23 @@ const (
|
||||
KindFeature = "feature"
|
||||
)
|
||||
|
||||
// Component scopes.
|
||||
//
|
||||
// A component is priced by one Paddle product, and how many catalogue rows it
|
||||
// needs follows from how many products it is. The base fee is a different
|
||||
// product per plan, so it is a row per plan. Every add-on — the server limit and
|
||||
// all four features — is ONE product sold to every paid plan at one price, so it
|
||||
// is one row, and its price ID is typed once instead of four times.
|
||||
//
|
||||
// Scope is stored rather than inferred from Kind so the rule is data. Pricing a
|
||||
// future add-on per tier is then a scope on a row, not a rewrite of every reader.
|
||||
const (
|
||||
// ScopePlan rows carry a deployment and a tier and belong to that plan alone.
|
||||
ScopePlan = "plan"
|
||||
// ScopeShared rows leave deployment and tier empty and belong to every paid plan.
|
||||
ScopeShared = "shared"
|
||||
)
|
||||
|
||||
// LimitKeyServers is the only metered limit today.
|
||||
//
|
||||
// A limit_key is a field name in license.Limits, which is what lets a second
|
||||
@@ -27,19 +46,23 @@ const (
|
||||
// would be 1 in every row that will ever exist.
|
||||
const LimitKeyServers = "max_servers"
|
||||
|
||||
// CatalogueRow is one priceable component of one plan.
|
||||
// CatalogueRow is one priceable component.
|
||||
//
|
||||
// This is the ONLY place a Paddle price ID appears anywhere in Vantage. An empty
|
||||
// PriceIDs means the component is free — a feature with no price is a toggle a
|
||||
// customer may take at no charge, and giving it a price later is a staff edit
|
||||
// rather than a migration or a deploy.
|
||||
type CatalogueRow struct {
|
||||
ID bson.ObjectID `bson:"_id,omitempty" json:"-"`
|
||||
Kind string `bson:"kind" json:"kind"`
|
||||
Deployment string `bson:"deployment" json:"deployment"`
|
||||
Tier string `bson:"tier" json:"tier"`
|
||||
LimitKey string `bson:"limit_key,omitempty" json:"limit_key,omitempty"`
|
||||
FeatureKey string `bson:"feature_key,omitempty" json:"feature_key,omitempty"`
|
||||
ID bson.ObjectID `bson:"_id,omitempty" json:"-"`
|
||||
Kind string `bson:"kind" json:"kind"`
|
||||
Scope string `bson:"scope" json:"scope"`
|
||||
// Deployment and Tier are empty on a shared row, and are what a plan row is
|
||||
// keyed by. Readers must go through CatalogueFor rather than filtering on
|
||||
// them, or a shared row is invisible to the plan that sells it.
|
||||
Deployment string `bson:"deployment" json:"deployment"`
|
||||
Tier string `bson:"tier" json:"tier"`
|
||||
LimitKey string `bson:"limit_key,omitempty" json:"limit_key,omitempty"`
|
||||
FeatureKey string `bson:"feature_key,omitempty" json:"feature_key,omitempty"`
|
||||
// PriceIDs is environment -> term -> Paddle price ID, e.g.
|
||||
// {"sandbox": {"monthly": "pri_…"}, "production": {"annual": "pri_…"}}.
|
||||
//
|
||||
@@ -67,57 +90,174 @@ func (r CatalogueRow) Priced(env string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// SeedCatalogue inserts the twenty rows the four PAID plans need: a base, a
|
||||
// server limit, and one row per feature key.
|
||||
// Shared reports whether this row is sold by every paid plan.
|
||||
func (r CatalogueRow) Shared() bool { return r.Scope == ScopeShared }
|
||||
|
||||
// naturalKey is how a row is addressed everywhere: by what it is, never by its
|
||||
// ObjectID. A shared row's deployment and tier are empty, and that emptiness is
|
||||
// part of the key rather than a wildcard.
|
||||
func (r CatalogueRow) naturalKey() bson.M {
|
||||
return bson.M{
|
||||
"kind": r.Kind,
|
||||
"deployment": r.Deployment,
|
||||
"tier": r.Tier,
|
||||
"limit_key": r.LimitKey,
|
||||
"feature_key": r.FeatureKey,
|
||||
}
|
||||
}
|
||||
|
||||
// seedRows is the catalogue as it should exist: four base rows, one per paid
|
||||
// plan, plus five shared add-on rows every paid plan sells.
|
||||
//
|
||||
// Nine rows, down from twenty-four. The count moves whenever shared/license
|
||||
// gains a feature, and this comment is how the next person knows the number was
|
||||
// chosen rather than drifted.
|
||||
//
|
||||
// The two Free plans get no rows at all, and that absence is what keeps Free
|
||||
// outside Paddle: with nothing to price, no checkout can be built for it. Do not
|
||||
// "fix" this by adding zero-priced Free rows.
|
||||
func seedRows() []CatalogueRow {
|
||||
rows := []CatalogueRow{}
|
||||
paid := []string{license.TierProfessional, license.TierEnterprise}
|
||||
for _, deployment := range license.Deployments() {
|
||||
for _, tier := range paid {
|
||||
rows = append(rows, CatalogueRow{
|
||||
Kind: KindBase, Scope: ScopePlan, Deployment: deployment, Tier: tier,
|
||||
})
|
||||
}
|
||||
}
|
||||
rows = append(rows, CatalogueRow{
|
||||
Kind: KindLimit, Scope: ScopeShared, LimitKey: LimitKeyServers,
|
||||
})
|
||||
for _, f := range []string{
|
||||
license.FeatureConsole,
|
||||
license.FeatureOIDC,
|
||||
license.FeatureVulnScanning,
|
||||
license.FeatureStatusPages,
|
||||
} {
|
||||
rows = append(rows, CatalogueRow{
|
||||
Kind: KindFeature, Scope: ScopeShared, FeatureKey: f,
|
||||
})
|
||||
}
|
||||
return rows
|
||||
}
|
||||
|
||||
// SeedCatalogue inserts the nine rows the four paid plans need.
|
||||
//
|
||||
// $setOnInsert only, for the same reason as SeedPlans: the price IDs are pasted
|
||||
// in by staff and a redeploy must not blank them.
|
||||
func SeedCatalogue(ctx context.Context) error {
|
||||
paid := []string{license.TierProfessional, license.TierEnterprise}
|
||||
for _, deployment := range license.Deployments() {
|
||||
for _, tier := range paid {
|
||||
rows := []CatalogueRow{
|
||||
{Kind: KindBase, Deployment: deployment, Tier: tier},
|
||||
{Kind: KindLimit, Deployment: deployment, Tier: tier, LimitKey: LimitKeyServers},
|
||||
{Kind: KindFeature, Deployment: deployment, Tier: tier, FeatureKey: license.FeatureConsole},
|
||||
{Kind: KindFeature, Deployment: deployment, Tier: tier, FeatureKey: license.FeatureOIDC},
|
||||
{Kind: KindFeature, Deployment: deployment, Tier: tier, FeatureKey: license.FeatureVulnScanning},
|
||||
}
|
||||
for _, r := range rows {
|
||||
filter := bson.M{
|
||||
"kind": r.Kind,
|
||||
"deployment": r.Deployment,
|
||||
"tier": r.Tier,
|
||||
"limit_key": r.LimitKey,
|
||||
"feature_key": r.FeatureKey,
|
||||
}
|
||||
if _, err := db.Admin("catalogue").UpdateOne(ctx, filter,
|
||||
bson.M{"$setOnInsert": bson.M{
|
||||
"kind": r.Kind,
|
||||
"deployment": r.Deployment,
|
||||
"tier": r.Tier,
|
||||
"limit_key": r.LimitKey,
|
||||
"feature_key": r.FeatureKey,
|
||||
"price_ids": map[string]map[string]string{},
|
||||
}},
|
||||
options.UpdateOne().SetUpsert(true)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
for _, r := range seedRows() {
|
||||
set := r.naturalKey()
|
||||
set["scope"] = r.Scope
|
||||
set["price_ids"] = map[string]map[string]string{}
|
||||
if _, err := db.Admin("catalogue").UpdateOne(ctx, r.naturalKey(),
|
||||
bson.M{"$setOnInsert": set},
|
||||
options.UpdateOne().SetUpsert(true)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// CatalogueFor returns every component of one plan.
|
||||
// MigrateSharedCatalogue collapses the four per-plan copies of each add-on onto
|
||||
// the one shared row, and deletes the copies.
|
||||
//
|
||||
// It runs after SeedCatalogue, which has already created the shared rows empty,
|
||||
// and is idempotent: once the per-plan copies are gone there is nothing to move.
|
||||
//
|
||||
// It REFUSES rather than guesses when the copies disagree. Four rows that were
|
||||
// meant to be one price and are not is a real pricing decision somebody made,
|
||||
// and picking one of them silently would move a customer's bill.
|
||||
func MigrateSharedCatalogue(ctx context.Context) error {
|
||||
// Rows seeded before scope existed are all per-plan rows. Naming them so
|
||||
// keeps CatalogueFor's $or honest for the base rows that survive.
|
||||
if _, err := db.Admin("catalogue").UpdateMany(ctx,
|
||||
bson.M{"scope": bson.M{"$exists": false}},
|
||||
bson.M{"$set": bson.M{"scope": ScopePlan}}); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
for _, shared := range seedRows() {
|
||||
if !shared.Shared() {
|
||||
continue
|
||||
}
|
||||
cur, err := db.Admin("catalogue").Find(ctx, bson.M{
|
||||
"kind": shared.Kind,
|
||||
"limit_key": shared.LimitKey,
|
||||
"feature_key": shared.FeatureKey,
|
||||
"deployment": bson.M{"$ne": ""},
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
old := []CatalogueRow{}
|
||||
if err := cur.All(ctx, &old); err != nil {
|
||||
return err
|
||||
}
|
||||
if len(old) == 0 {
|
||||
continue
|
||||
}
|
||||
|
||||
var target CatalogueRow
|
||||
if err := db.Admin("catalogue").FindOne(ctx, shared.naturalKey()).Decode(&target); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
merged := target.PriceIDs
|
||||
if merged == nil {
|
||||
merged = map[string]map[string]string{}
|
||||
}
|
||||
for _, o := range old {
|
||||
for env, byTerm := range o.PriceIDs {
|
||||
for term, id := range byTerm {
|
||||
if id == "" {
|
||||
continue
|
||||
}
|
||||
if merged[env] == nil {
|
||||
merged[env] = map[string]string{}
|
||||
}
|
||||
if have := merged[env][term]; have != "" && have != id {
|
||||
return fmt.Errorf(
|
||||
"catalogue: %s%s was priced differently per plan (%s %s: %q and %q); "+
|
||||
"decide which price is the shared one and delete the others before upgrading",
|
||||
shared.LimitKey, shared.FeatureKey, env, term, have, id)
|
||||
}
|
||||
merged[env][term] = id
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := db.Admin("catalogue").UpdateOne(ctx, shared.naturalKey(),
|
||||
bson.M{"$set": bson.M{"price_ids": merged}}); err != nil {
|
||||
return err
|
||||
}
|
||||
ids := make([]bson.ObjectID, 0, len(old))
|
||||
for _, o := range old {
|
||||
ids = append(ids, o.ID)
|
||||
}
|
||||
if _, err := db.Admin("catalogue").DeleteMany(ctx,
|
||||
bson.M{"_id": bson.M{"$in": ids}}); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Printf("catalogue: merged %d per-plan rows into shared %s%s",
|
||||
len(old), shared.LimitKey, shared.FeatureKey)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// CatalogueFor returns every component one plan sells: its own base row plus
|
||||
// every shared add-on.
|
||||
//
|
||||
// This is the seam the whole shared-row change rests on. Every reader that used
|
||||
// to filter the catalogue by deployment and tier must come through here instead,
|
||||
// or it sees a plan priced by nothing but its base fee.
|
||||
func CatalogueFor(ctx context.Context, deployment, tier string) ([]CatalogueRow, error) {
|
||||
deployment, tier = license.NormaliseTier(deployment, tier)
|
||||
cur, err := db.Admin("catalogue").Find(ctx,
|
||||
bson.M{"deployment": deployment, "tier": tier})
|
||||
cur, err := db.Admin("catalogue").Find(ctx, bson.M{"$or": []bson.M{
|
||||
{"scope": ScopeShared},
|
||||
{"deployment": deployment, "tier": tier},
|
||||
}})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import { useEffect, useMemo, useState } from "react";
|
||||
import { useRouter } from "next/navigation";
|
||||
import Link from "next/link";
|
||||
import { useMutation, useQuery } from "@tanstack/react-query";
|
||||
import { rowsForPlan, sharedRows } from "@/lib/catalogue";
|
||||
import { ApiError, api, lineItemsFor, type CatalogueRow, type CheckoutOptions, type Deployment, type Plan, type Term, type Tier } from "@/lib/api";
|
||||
import { initPaddle, previewPrices, type PricePreview } from "@/lib/paddle";
|
||||
import { featureDesc, featureLabel } from "@/lib/features";
|
||||
@@ -60,24 +61,24 @@ export function PurchaseForm() {
|
||||
const options = optionsQ.data;
|
||||
const accountId = account.data?.account.account_id ?? "";
|
||||
|
||||
// Distinct feature keys offered on this deployment, in a stable order.
|
||||
// Every feature a paid plan can be sold, in a stable order. Features are
|
||||
// shared rows now, so they no longer differ by deployment — the list is the
|
||||
// same on both, and reads from one place rather than four.
|
||||
const featureKeys = useMemo(() => {
|
||||
if (!options) return [] as string[];
|
||||
const keys = new Set<string>();
|
||||
for (const r of options.catalogue) {
|
||||
if (r.deployment === dep && r.kind === "feature" && r.feature_key) {
|
||||
keys.add(r.feature_key);
|
||||
}
|
||||
for (const r of sharedRows(options.catalogue)) {
|
||||
if (r.kind === "feature" && r.feature_key) keys.add(r.feature_key);
|
||||
}
|
||||
return [...keys];
|
||||
}, [options, dep]);
|
||||
}, [options]);
|
||||
|
||||
const activePlans = useMemo(() => (options?.plans ?? []).filter((p) => p.deployment === dep && p.active).sort((a, b) => TIER_ORDER.indexOf(a.tier) - TIER_ORDER.indexOf(b.tier)), [options, dep]);
|
||||
const plan = activePlans.find((p) => p.tier === choice.tier);
|
||||
const baseServers = plan?.base_limits.max_servers ?? 0;
|
||||
const unlimited = baseServers === -1;
|
||||
|
||||
const rows = useMemo(() => (options?.catalogue ?? []).filter((r) => r.deployment === dep && r.tier === choice.tier), [options, dep, choice.tier]);
|
||||
const rows = useMemo(() => rowsForPlan(options?.catalogue ?? [], dep, choice.tier), [options, dep, choice.tier]);
|
||||
|
||||
// Real line items for the current configuration the same builder the
|
||||
// checkout uses, so the summary can never disagree with the overlay.
|
||||
@@ -239,7 +240,7 @@ export function PurchaseForm() {
|
||||
headline={p.tier === "free" ? "£0" : basePrices[p.tier]}
|
||||
cycleLabel={cycleShort(dep, choice.term)}
|
||||
featureKeys={featureKeys}
|
||||
catalogue={options.catalogue.filter((r) => r.deployment === dep && r.tier === p.tier)}
|
||||
catalogue={rowsForPlan(options.catalogue, dep, p.tier)}
|
||||
env={options.env}
|
||||
term={choice.term}
|
||||
onSelect={() =>
|
||||
@@ -252,7 +253,7 @@ export function PurchaseForm() {
|
||||
features: c.features.filter((k) => {
|
||||
const st = featureStateFor(
|
||||
p,
|
||||
options.catalogue.filter((r) => r.deployment === dep && r.tier === p.tier),
|
||||
rowsForPlan(options.catalogue, dep, p.tier),
|
||||
options.env,
|
||||
c.term,
|
||||
k,
|
||||
@@ -658,7 +659,7 @@ function Receipt({
|
||||
// Label each real line item from the catalogue, and price it from Paddle.
|
||||
const base = plan?.base_limits.max_servers ?? 0;
|
||||
const extra = base === -1 ? 0 : Math.max(0, choice.servers - base);
|
||||
const rows = options.catalogue.filter((r) => r.deployment === dep && r.tier === choice.tier);
|
||||
const rows = rowsForPlan(options.catalogue, dep, choice.tier);
|
||||
const idFor = (predicate: (r: CatalogueRow) => boolean) => {
|
||||
const row = rows.find(predicate);
|
||||
return row?.price_ids?.[options.env]?.[choice.term] ?? "";
|
||||
|
||||
@@ -1,169 +0,0 @@
|
||||
"use client";
|
||||
|
||||
import { useState } from "react";
|
||||
import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query";
|
||||
import { PageHeader } from "@/components/PageHeader";
|
||||
import { PageFrame } from "@/components/PageFrame";
|
||||
import { Panel } from "@/components/Panel";
|
||||
import { TBody, TD, TH, THead, TR, Table } from "@/components/Table";
|
||||
import { api, type CatalogueRow, type Term } from "@/lib/api";
|
||||
|
||||
const ENVS = ["sandbox", "production"] as const;
|
||||
|
||||
/* Self-hosted sells annual only, so the monthly cell is not rendered for it
|
||||
* rather than rendered and rejected. The backend refuses one either way; this is
|
||||
* so nobody types into a field that cannot be saved. */
|
||||
function termsFor(deployment: string): Term[] {
|
||||
return deployment === "self_hosted" ? ["annual"] : ["monthly", "annual"];
|
||||
}
|
||||
|
||||
function componentLabel(r: CatalogueRow): string {
|
||||
if (r.kind === "base") return "Base fee";
|
||||
if (r.kind === "limit") return `Per ${r.limit_key?.replace("max_", "")}`;
|
||||
return `Feature: ${r.feature_key}`;
|
||||
}
|
||||
|
||||
function rowKey(r: CatalogueRow): string {
|
||||
return [r.deployment, r.tier, r.kind, r.limit_key ?? "", r.feature_key ?? ""].join("/");
|
||||
}
|
||||
|
||||
export default function CataloguePage() {
|
||||
const qc = useQueryClient();
|
||||
const { data: rows = [], isLoading } = useQuery({
|
||||
queryKey: ["staff", "catalogue"],
|
||||
queryFn: api.staff.catalogue,
|
||||
});
|
||||
const [drafts, setDrafts] = useState<Record<string, CatalogueRow["price_ids"]>>({});
|
||||
|
||||
const save = useMutation({
|
||||
mutationFn: (r: CatalogueRow) => api.staff.updateCatalogue(r),
|
||||
onSuccess: () => qc.invalidateQueries({ queryKey: ["staff", "catalogue"] }),
|
||||
});
|
||||
|
||||
const groups = Array.from(new Set(rows.map((r) => `${r.deployment}/${r.tier}`)));
|
||||
|
||||
return (
|
||||
<div className="grid gap-6">
|
||||
<PageHeader
|
||||
title="Catalogue"
|
||||
back={{ href: "/staff", label: "Operations" }}
|
||||
subtitle="Every priceable component. This is the only place a Paddle price ID lives."
|
||||
/>
|
||||
<PageFrame
|
||||
aside={
|
||||
<aside className="space-y-3 text-[0.82rem] text-ink-2">
|
||||
<p>
|
||||
A component with no price ID is free. A feature with no price is a
|
||||
toggle a customer may take at no charge; giving it a price here is
|
||||
all it takes to start charging for it.
|
||||
</p>
|
||||
<p>
|
||||
Free is priced by nothing and has no rows. That absence is what
|
||||
keeps it outside Paddle.
|
||||
</p>
|
||||
<p>
|
||||
Changing a price affects the next checkout only. It cannot touch an
|
||||
issued licence.
|
||||
</p>
|
||||
</aside>
|
||||
}
|
||||
>
|
||||
{isLoading ? (
|
||||
<p className="text-[0.85rem] text-ink-3">Loading…</p>
|
||||
) : (
|
||||
<div className="grid gap-4">
|
||||
{groups.map((g) => {
|
||||
const [deployment, tier] = g.split("/");
|
||||
const terms = termsFor(deployment);
|
||||
return (
|
||||
<Panel key={g} title={`${deployment === "cloud" ? "Cloud" : "Self-Hosted"} ${tier}`} meta={terms.join(" · ")} bodyless>
|
||||
<Table className="min-w-[42rem]">
|
||||
<THead>
|
||||
<TR className="hover:bg-transparent">
|
||||
<TH>Component</TH>
|
||||
{ENVS.map((env) =>
|
||||
terms.map((t) => (
|
||||
<TH key={`${env}-${t}`}>
|
||||
{env} / {t}
|
||||
</TH>
|
||||
)),
|
||||
)}
|
||||
<TH />
|
||||
</TR>
|
||||
</THead>
|
||||
<TBody>
|
||||
{rows
|
||||
.filter(
|
||||
(r) =>
|
||||
r.deployment === deployment &&
|
||||
r.tier === tier,
|
||||
)
|
||||
.map((r) => {
|
||||
const k = rowKey(r);
|
||||
const ids = drafts[k] ?? r.price_ids ?? {};
|
||||
const dirty =
|
||||
JSON.stringify(ids) !==
|
||||
JSON.stringify(r.price_ids ?? {});
|
||||
return (
|
||||
<TR key={k}>
|
||||
<TD className="text-ink">{componentLabel(r)}</TD>
|
||||
{ENVS.map((env) =>
|
||||
terms.map((t) => (
|
||||
<TD key={`${env}-${t}`}>
|
||||
<input
|
||||
value={
|
||||
ids[env]?.[t] ?? ""
|
||||
}
|
||||
placeholder="pri_…"
|
||||
onChange={(e) =>
|
||||
setDrafts({
|
||||
...drafts,
|
||||
[k]: {
|
||||
...ids,
|
||||
[env]: {
|
||||
...(ids[
|
||||
env
|
||||
] ?? {}),
|
||||
[t]: e
|
||||
.target
|
||||
.value,
|
||||
},
|
||||
},
|
||||
})
|
||||
}
|
||||
className="w-40 rounded border border-rule bg-panel-2 px-2 py-1 font-mono text-[0.78rem] text-ink focus:border-accent focus:outline-none"
|
||||
/>
|
||||
</TD>
|
||||
)),
|
||||
)}
|
||||
<TD numeric>
|
||||
<button
|
||||
type="button"
|
||||
disabled={
|
||||
!dirty || save.isPending
|
||||
}
|
||||
onClick={() =>
|
||||
save.mutate({
|
||||
...r,
|
||||
price_ids: ids,
|
||||
})
|
||||
}
|
||||
className="rounded border border-accent px-2.5 py-1 font-mono text-[0.7rem] uppercase tracking-[0.1em] text-accent disabled:opacity-40"
|
||||
>
|
||||
Save
|
||||
</button>
|
||||
</TD>
|
||||
</TR>
|
||||
);
|
||||
})}
|
||||
</TBody>
|
||||
</Table>
|
||||
</Panel>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
)}
|
||||
</PageFrame>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -7,8 +7,7 @@ const LINKS: NavLink[] = [
|
||||
{ href: "/staff", label: "Operations" },
|
||||
{ href: "/staff/accounts", label: "Accounts" },
|
||||
{ href: "/staff/licenses", label: "Licences" },
|
||||
{ href: "/staff/plans", label: "Plans" },
|
||||
{ href: "/staff/catalogue", label: "Catalogue" },
|
||||
{ href: "/staff/pricing", label: "Pricing" },
|
||||
{ href: "/staff/audit", label: "Audit" },
|
||||
];
|
||||
|
||||
|
||||
@@ -1,153 +0,0 @@
|
||||
"use client";
|
||||
|
||||
import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query";
|
||||
import { useState } from "react";
|
||||
import { api, type Deployment, type Plan, type Tier } from "@/lib/api";
|
||||
import { ConfirmPlanChange } from "@/components/ConfirmPlanChange";
|
||||
import { PageHeader } from "@/components/PageHeader";
|
||||
import { Panel } from "@/components/Panel";
|
||||
|
||||
const SUPPORT_LEVELS = [
|
||||
{ value: "community", label: "Community" },
|
||||
{ value: "email_24_5", label: "Email, 24/5" },
|
||||
{ value: "email_call_24_7", label: "Email + call, 24/7" },
|
||||
] as const;
|
||||
|
||||
const LIMIT_FIELDS = [
|
||||
{ key: "max_servers", label: "Servers" },
|
||||
{ key: "max_monitors", label: "Monitors" },
|
||||
{ key: "max_secret_groups", label: "Secret groups" },
|
||||
{ key: "max_channels", label: "Channels" },
|
||||
{ key: "audit_retention_days", label: "Audit history (days)" },
|
||||
] as const;
|
||||
|
||||
/*
|
||||
* -1 is Unlimited everywhere in the licence payload, so the form takes it
|
||||
* literally rather than inventing a checkbox. A staff screen that hides the
|
||||
* sentinel is a staff screen where nobody can tell whether a plan says
|
||||
* unlimited or nothing at all.
|
||||
*/
|
||||
function AllowanceForm({ plan, onSave, saving }: { plan: Plan; onSave: (next: Plan) => void; saving: boolean }) {
|
||||
const [draft, setDraft] = useState<Plan>(plan);
|
||||
const dirty = JSON.stringify(draft) !== JSON.stringify(plan);
|
||||
|
||||
return (
|
||||
<div className="grid gap-3">
|
||||
<div className="grid gap-2 sm:grid-cols-2 lg:grid-cols-3">
|
||||
{LIMIT_FIELDS.map((f) => (
|
||||
<label key={f.key} className="block">
|
||||
<span className="mb-1 block text-[0.78rem] text-ink-3">{f.label}</span>
|
||||
<input
|
||||
type="number"
|
||||
value={draft.base_limits[f.key]}
|
||||
onChange={(e) =>
|
||||
setDraft({
|
||||
...draft,
|
||||
base_limits: {
|
||||
...draft.base_limits,
|
||||
[f.key]: Number(e.target.value),
|
||||
},
|
||||
})
|
||||
}
|
||||
className="w-full rounded border border-rule bg-panel-2 px-2 py-1.5 text-[0.85rem] text-ink focus:border-accent focus:outline-none"
|
||||
/>
|
||||
<span className="mt-0.5 block text-[0.72rem] text-ink-3">−1 is unlimited</span>
|
||||
</label>
|
||||
))}
|
||||
<label className="block">
|
||||
<span className="mb-1 block text-[0.78rem] text-ink-3">Support level</span>
|
||||
<select
|
||||
value={draft.support_level}
|
||||
onChange={(e) => setDraft({ ...draft, support_level: e.target.value })}
|
||||
className="w-full rounded border border-rule bg-panel-2 px-2 py-1.5 text-[0.85rem] text-ink focus:border-accent focus:outline-none"
|
||||
>
|
||||
{SUPPORT_LEVELS.map((s) => (
|
||||
<option key={s.value} value={s.value}>
|
||||
{s.label}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
</label>
|
||||
</div>
|
||||
|
||||
<label className="flex items-center gap-2 text-[0.85rem] text-ink-2">
|
||||
<input type="checkbox" checked={draft.active} onChange={(e) => setDraft({ ...draft, active: e.target.checked })} />
|
||||
Offered to customers
|
||||
</label>
|
||||
|
||||
<p className="text-[0.78rem] text-ink-3">Changes apply to licences issued from now on. Existing licences snapshotted their plan and are unaffected.</p>
|
||||
|
||||
<button
|
||||
type="button"
|
||||
disabled={!dirty || saving}
|
||||
onClick={() => onSave(draft)}
|
||||
className="justify-self-start rounded border border-accent bg-accent px-3.5 py-2 text-[0.86rem] font-semibold text-accent-ink disabled:opacity-40"
|
||||
>
|
||||
{saving ? "Saving…" : "Save allowances"}
|
||||
</button>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
export default function PlansPage() {
|
||||
const qc = useQueryClient();
|
||||
const plans = useQuery({ queryKey: ["plans"], queryFn: api.staff.plans });
|
||||
const licenses = useQuery({
|
||||
queryKey: ["staff-licenses"],
|
||||
queryFn: () => api.staff.licenses(),
|
||||
});
|
||||
const [draft, setDraft] = useState<Plan | null>(null);
|
||||
const [saving, setSaving] = useState<string | null>(null);
|
||||
|
||||
const save = useMutation({
|
||||
mutationFn: (p: Plan) => api.staff.updatePlan(p.deployment, p.tier, p),
|
||||
onSuccess: () => {
|
||||
qc.invalidateQueries({ queryKey: ["plans"] });
|
||||
setDraft(null);
|
||||
setSaving(null);
|
||||
},
|
||||
onError: () => setSaving(null),
|
||||
});
|
||||
|
||||
const original = plans.data?.find((p) => p.deployment === draft?.deployment && p.tier === draft?.tier);
|
||||
|
||||
return (
|
||||
<div className="grid gap-6">
|
||||
<PageHeader
|
||||
title="Plans"
|
||||
subtitle="The authoritative tier table six plans, two deployments by three tiers, base allowances only. Every issued licence snapshots the plan it was cut from, so editing one never rewrites an existing licence."
|
||||
/>
|
||||
|
||||
{draft && original && (
|
||||
<ConfirmPlanChange
|
||||
plan={original}
|
||||
next={draft}
|
||||
issuedCount={(licenses.data ?? []).filter((l) => l.tier === draft.tier && l.deployment === draft.deployment).length}
|
||||
onConfirm={() => {
|
||||
setSaving(`${draft.deployment}/${draft.tier}`);
|
||||
save.mutate(draft);
|
||||
}}
|
||||
onCancel={() => setDraft(null)}
|
||||
/>
|
||||
)}
|
||||
|
||||
{(["cloud", "self_hosted"] as const).map((deployment: Deployment) => (
|
||||
<section key={deployment} className="grid gap-3">
|
||||
<h2 className="font-mono text-[0.68rem] uppercase tracking-[0.14em] text-ink-3">{deployment === "cloud" ? "Cloud" : "Self-Hosted"}</h2>
|
||||
{(plans.data ?? [])
|
||||
.filter((p) => p.deployment === deployment)
|
||||
.map((p) => (
|
||||
<Panel
|
||||
key={`${p.deployment}/${p.tier}`}
|
||||
title={p.name}
|
||||
meta={`${p.deployment}/${p.tier}`}
|
||||
actions={!p.active ? <span className="font-mono text-[0.64rem] uppercase tracking-[0.12em] text-warn">Not offered</span> : undefined}
|
||||
>
|
||||
<AllowanceForm plan={p} saving={saving === `${p.deployment}/${p.tier}`} onSave={(next: Plan) => setDraft(next)} />
|
||||
</Panel>
|
||||
))}
|
||||
</section>
|
||||
))}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,162 @@
|
||||
"use client";
|
||||
|
||||
import { useState } from "react";
|
||||
import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query";
|
||||
import { Panel } from "@/components/Panel";
|
||||
import { SectionHeading } from "./SectionHeading";
|
||||
import { planRows, rowKey, sharedRows } from "@/lib/catalogue";
|
||||
import { featureLabel } from "@/lib/features";
|
||||
import { api, type CatalogueRow, type Term } from "@/lib/api";
|
||||
|
||||
const ENVS = ["sandbox", "production"] as const;
|
||||
|
||||
/* A shared row is sold by both deployments, so it holds both terms: the cloud
|
||||
* checkout takes the monthly price and the self-hosted one never asks for it. A
|
||||
* plan row offers only the terms its own deployment sells — self-hosted is
|
||||
* annual only, and the field is not rendered rather than rendered and refused. */
|
||||
function termsFor(r: CatalogueRow): Term[] {
|
||||
if (r.scope === "shared") return ["monthly", "annual"];
|
||||
return r.deployment === "self_hosted" ? ["annual"] : ["monthly", "annual"];
|
||||
}
|
||||
|
||||
function componentLabel(r: CatalogueRow): string {
|
||||
if (r.kind === "base") return `${r.tier === "enterprise" ? "Enterprise" : "Professional"} (${r.deployment === "cloud" ? "Cloud" : "Self-hosted"})`;
|
||||
if (r.kind === "limit") return "Additional server";
|
||||
return featureLabel(r.feature_key ?? "");
|
||||
}
|
||||
|
||||
function componentDetail(r: CatalogueRow): string {
|
||||
if (r.kind === "base") return "The plan's own fee, always quantity 1";
|
||||
if (r.kind === "limit") return `Raises ${r.limit_key} by one per unit`;
|
||||
return `feature · ${r.feature_key}`;
|
||||
}
|
||||
|
||||
/*
|
||||
* The coverage ledger: one square per environment and term, filled when that
|
||||
* cell holds a price ID.
|
||||
*
|
||||
* A missing production price is invisible in a grid of text inputs — every cell
|
||||
* looks like every other until you read twenty-six characters of each. This is
|
||||
* the one thing staff come to this page to check before a launch, so it reads
|
||||
* before the IDs do.
|
||||
*/
|
||||
function Coverage({ row, terms }: { row: CatalogueRow; terms: Term[] }) {
|
||||
const cells = ENVS.flatMap((env) => terms.map((t) => ({ env, t, filled: Boolean(row.price_ids?.[env]?.[t]) })));
|
||||
const filled = cells.filter((c) => c.filled).length;
|
||||
return (
|
||||
<span className="flex items-center gap-1">
|
||||
{cells.map((c) => (
|
||||
<span key={`${c.env}-${c.t}`} title={`${c.env} ${c.t}`} className={["block h-2.5 w-2.5 rounded-[1px] border", c.filled ? "border-valid bg-valid" : "border-rule bg-panel-2"].join(" ")} />
|
||||
))}
|
||||
<span className="ml-1.5 font-mono text-[0.62rem] tracking-[0.08em] text-ink-3">
|
||||
{filled}/{cells.length} priced
|
||||
</span>
|
||||
</span>
|
||||
);
|
||||
}
|
||||
|
||||
function ComponentRow({ row, scopeLabel }: { row: CatalogueRow; scopeLabel: string }) {
|
||||
const qc = useQueryClient();
|
||||
const [draft, setDraft] = useState<CatalogueRow["price_ids"] | null>(null);
|
||||
const ids = draft ?? row.price_ids ?? {};
|
||||
const dirty = JSON.stringify(ids) !== JSON.stringify(row.price_ids ?? {});
|
||||
const terms = termsFor(row);
|
||||
|
||||
const save = useMutation({
|
||||
mutationFn: () => api.staff.updateCatalogue({ ...row, price_ids: ids }),
|
||||
onSuccess: () => {
|
||||
setDraft(null);
|
||||
qc.invalidateQueries({ queryKey: ["staff", "catalogue"] });
|
||||
},
|
||||
});
|
||||
|
||||
const set = (env: string, term: Term, value: string) =>
|
||||
setDraft({ ...ids, [env]: { ...(ids[env] ?? {}), [term]: value } });
|
||||
|
||||
return (
|
||||
<div className="grid gap-3 border-t border-rule-soft pt-3 first:border-0 first:pt-0 md:grid-cols-[minmax(0,17rem)_1fr]">
|
||||
<div className="grid content-start gap-1.5">
|
||||
<span className="text-[0.9rem] font-semibold">{componentLabel(row)}</span>
|
||||
<span className={["w-max rounded border px-1.5 py-px font-mono text-[0.6rem] uppercase tracking-[0.1em]", row.scope === "shared" ? "border-accent text-accent" : "border-rule text-ink-3"].join(" ")}>{scopeLabel}</span>
|
||||
<span className="text-[0.78rem] text-ink-3">{componentDetail(row)}</span>
|
||||
<Coverage row={{ ...row, price_ids: ids }} terms={terms} />
|
||||
</div>
|
||||
|
||||
<div className="grid gap-2">
|
||||
<div className="grid gap-1.5 sm:grid-cols-2">
|
||||
{ENVS.map((env) => (
|
||||
<div key={env} className="grid content-start gap-1.5">
|
||||
<span className="flex items-center gap-2 font-mono text-[0.62rem] uppercase tracking-[0.12em] text-ink-3">
|
||||
{env}
|
||||
<span className="h-px flex-1 bg-rule-soft" />
|
||||
</span>
|
||||
{terms.map((t) => (
|
||||
<label key={t} className="grid gap-1">
|
||||
<span className="font-mono text-[0.62rem] uppercase tracking-[0.1em] text-ink-3">{t}</span>
|
||||
<input
|
||||
value={ids[env]?.[t] ?? ""}
|
||||
placeholder="pri_…"
|
||||
onChange={(e) => set(env, t, e.target.value)}
|
||||
className={["w-full rounded border bg-panel-2 px-2 py-1.5 font-mono text-[0.76rem] text-ink focus:border-accent focus:outline-none", ids[env]?.[t] ? "border-rule" : "border-dashed border-rule"].join(" ")}
|
||||
aria-label={`${componentLabel(row)} ${env} ${t} price ID`}
|
||||
/>
|
||||
</label>
|
||||
))}
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
<div className="flex flex-wrap items-center gap-2.5">
|
||||
<button type="button" disabled={!dirty || save.isPending} onClick={() => save.mutate()} className="rounded border border-accent px-2.5 py-1 font-mono text-[0.7rem] uppercase tracking-[0.1em] text-accent disabled:opacity-40">
|
||||
{save.isPending ? "Saving…" : "Save"}
|
||||
</button>
|
||||
{save.error && <span className="text-[0.78rem] text-expired">{(save.error as Error).message}</span>}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/*
|
||||
* The catalogue half of /staff/pricing: every priceable component, grouped by
|
||||
* what it is rather than by which plan sells it.
|
||||
*/
|
||||
export function CatalogueSection() {
|
||||
const { data: rows = [], isLoading } = useQuery({
|
||||
queryKey: ["staff", "catalogue"],
|
||||
queryFn: api.staff.catalogue,
|
||||
});
|
||||
|
||||
const shared = sharedRows(rows);
|
||||
const bases = planRows(rows);
|
||||
|
||||
return (
|
||||
<section className="grid gap-3">
|
||||
<SectionHeading
|
||||
title="Catalogue"
|
||||
note="Every priceable component, grouped by what it is rather than by which plan sells it. This is the only place a Paddle price ID lives."
|
||||
/>
|
||||
|
||||
<div className="grid gap-1.5 rounded border-l-2 border-accent bg-accent-wash px-3 py-2.5 text-[0.82rem] text-ink-2">
|
||||
<p>An add-on is one Paddle product sold to every paid plan, so its price is typed once. Only the base fee differs by plan, because only the base fee is a different product per plan.</p>
|
||||
<p>A component with no price ID is free — a feature with no price is a toggle a customer may take at no charge. Free is priced by nothing and has no rows at all, which is what keeps it outside Paddle. Changing a price affects the next checkout only; it cannot touch an issued licence.</p>
|
||||
</div>
|
||||
|
||||
{isLoading ? (
|
||||
<p className="text-[0.85rem] text-ink-3">Loading…</p>
|
||||
) : (
|
||||
<div className="grid gap-3">
|
||||
<Panel title="Add-ons" meta={`${shared.length} rows · every paid plan`}>
|
||||
{shared.map((r) => (
|
||||
<ComponentRow key={rowKey(r)} row={r} scopeLabel="All paid plans" />
|
||||
))}
|
||||
</Panel>
|
||||
<Panel title="Base fee" meta={`${bases.length} rows · one per plan`}>
|
||||
{bases.map((r) => (
|
||||
<ComponentRow key={rowKey(r)} row={r} scopeLabel="This plan only" />
|
||||
))}
|
||||
</Panel>
|
||||
</div>
|
||||
)}
|
||||
</section>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,248 @@
|
||||
"use client";
|
||||
|
||||
import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query";
|
||||
import { useState } from "react";
|
||||
import { api, type Deployment, type Plan } from "@/lib/api";
|
||||
import { featureDesc, featureLabel, FEATURE_LABEL } from "@/lib/features";
|
||||
import { limitLabel } from "@/lib/format";
|
||||
import { Button, controlClass } from "@/components/Button";
|
||||
import { ConfirmPlanChange } from "@/components/ConfirmPlanChange";
|
||||
import { SectionHeading } from "./SectionHeading";
|
||||
import { Modal } from "@/components/Modal";
|
||||
|
||||
const SUPPORT_LEVELS = [
|
||||
{ value: "community", label: "Community" },
|
||||
{ value: "email_24_5", label: "Email, 24/5" },
|
||||
{ value: "email_call_24_7", label: "Email + call, 24/7" },
|
||||
] as const;
|
||||
|
||||
const LIMIT_FIELDS = [
|
||||
{ key: "max_servers", label: "Servers" },
|
||||
{ key: "max_monitors", label: "Monitors" },
|
||||
{ key: "max_secret_groups", label: "Secret groups" },
|
||||
{ key: "max_channels", label: "Channels" },
|
||||
{ key: "audit_retention_days", label: "Audit history (days)" },
|
||||
] as const;
|
||||
|
||||
const FEATURE_KEYS = Object.keys(FEATURE_LABEL);
|
||||
|
||||
const planKey = (p: Plan) => `${p.deployment}/${p.tier}`;
|
||||
|
||||
/*
|
||||
* The list is tiers, and a tier's settings are behind a button.
|
||||
*
|
||||
* Six plans with five number fields, a select, a checkbox and four toggles each
|
||||
* is forty-odd controls on one screen, and the page it made could not be read
|
||||
* for the thing it exists to answer: what does each tier give you. The card
|
||||
* answers that; the modal is where it is changed.
|
||||
*/
|
||||
function TierCard({ plan, onOpen }: { plan: Plan; onOpen: () => void }) {
|
||||
return (
|
||||
<button
|
||||
type="button"
|
||||
onClick={onOpen}
|
||||
className={[
|
||||
"grid w-full gap-2.5 rounded border bg-panel p-3.5 text-left",
|
||||
"transition-[border-color,transform] duration-150 hover:-translate-y-px hover:border-accent",
|
||||
plan.active ? "border-rule" : "border-dashed border-rule opacity-75",
|
||||
].join(" ")}
|
||||
>
|
||||
<span className="flex flex-wrap items-center gap-2">
|
||||
<span className="text-[1rem] font-semibold">{plan.name}</span>
|
||||
{!plan.active && <span className="rounded border border-warn px-1.5 py-px font-mono text-[0.6rem] uppercase tracking-[0.1em] text-warn">Not offered</span>}
|
||||
<span className="ml-auto font-mono text-[0.68rem] text-ink-3">{planKey(plan)}</span>
|
||||
</span>
|
||||
|
||||
<dl className="grid grid-cols-[1fr_auto] gap-x-3 gap-y-0.5 text-[0.82rem]">
|
||||
<dt className="text-ink-3">Servers</dt>
|
||||
<dd className="text-right tabular-nums">{limitLabel(plan.base_limits.max_servers)}</dd>
|
||||
<dt className="text-ink-3">Monitors</dt>
|
||||
<dd className="text-right tabular-nums">{limitLabel(plan.base_limits.max_monitors)}</dd>
|
||||
<dt className="text-ink-3">Audit history</dt>
|
||||
<dd className="text-right tabular-nums">{limitLabel(plan.base_limits.audit_retention_days)} days</dd>
|
||||
</dl>
|
||||
|
||||
{/* Every feature key, lit or unlit — an absent chip cannot be told
|
||||
* from a feature nobody has heard of, and no tier bundles one today,
|
||||
* so the unlit row IS the information. */}
|
||||
<span className="flex flex-wrap gap-1">
|
||||
{FEATURE_KEYS.map((k) => {
|
||||
const on = plan.base_features.includes(k);
|
||||
return (
|
||||
<span key={k} className={["rounded border px-1.5 py-px font-mono text-[0.6rem] uppercase tracking-[0.06em]", on ? "border-valid text-valid" : "border-rule text-ink-3"].join(" ")}>
|
||||
{featureLabel(k)}
|
||||
</span>
|
||||
);
|
||||
})}
|
||||
</span>
|
||||
|
||||
<span className="justify-self-start rounded border border-accent px-2.5 py-1 font-mono text-[0.68rem] uppercase tracking-[0.1em] text-accent">Open plan</span>
|
||||
</button>
|
||||
);
|
||||
}
|
||||
|
||||
/* -1 is Unlimited everywhere in the licence payload, so the form takes it
|
||||
* literally rather than inventing a checkbox. A staff screen that hides the
|
||||
* sentinel is a staff screen where nobody can tell whether a plan says
|
||||
* unlimited or nothing at all. */
|
||||
function PlanModal({ plan, onClose, onSave }: { plan: Plan; onClose: () => void; onSave: (next: Plan) => void }) {
|
||||
const [draft, setDraft] = useState<Plan>(plan);
|
||||
const dirty = JSON.stringify(draft) !== JSON.stringify(plan);
|
||||
|
||||
const toggleFeature = (key: string, on: boolean) =>
|
||||
setDraft({
|
||||
...draft,
|
||||
base_features: on ? [...draft.base_features, key] : draft.base_features.filter((f) => f !== key),
|
||||
});
|
||||
|
||||
return (
|
||||
<Modal
|
||||
open
|
||||
onClose={onClose}
|
||||
title={plan.name}
|
||||
meta={planKey(plan)}
|
||||
footer={
|
||||
<>
|
||||
<p className="mr-auto max-w-md text-[0.78rem] text-ink-3">Applies to licences issued from now on. Issued licences snapshotted their plan and are unaffected.</p>
|
||||
<Button type="button" variant="line" onClick={onClose}>
|
||||
Cancel
|
||||
</Button>
|
||||
<Button type="button" disabled={!dirty} onClick={() => onSave(draft)}>
|
||||
Save plan
|
||||
</Button>
|
||||
</>
|
||||
}
|
||||
>
|
||||
<section className="grid gap-2">
|
||||
<span className="font-mono text-[0.66rem] uppercase tracking-[0.14em] text-ink-3">Base limits</span>
|
||||
<div className="grid gap-2 sm:grid-cols-2 lg:grid-cols-3">
|
||||
{LIMIT_FIELDS.map((f) => (
|
||||
<label key={f.key} className="grid gap-1">
|
||||
<span className="text-[0.78rem] text-ink-3">{f.label}</span>
|
||||
<input
|
||||
type="number"
|
||||
value={draft.base_limits[f.key]}
|
||||
onChange={(e) =>
|
||||
setDraft({
|
||||
...draft,
|
||||
base_limits: { ...draft.base_limits, [f.key]: Number(e.target.value) },
|
||||
})
|
||||
}
|
||||
className={controlClass("h-9 text-[0.84rem] tabular-nums")}
|
||||
/>
|
||||
</label>
|
||||
))}
|
||||
</div>
|
||||
<p className="text-[0.78rem] text-ink-3">−1 is unlimited. A metered dimension starts here and the customer buys upward from it.</p>
|
||||
</section>
|
||||
|
||||
<section className="grid gap-2">
|
||||
<span className="font-mono text-[0.66rem] uppercase tracking-[0.14em] text-ink-3">Base features</span>
|
||||
<div className="grid gap-1.5">
|
||||
{FEATURE_KEYS.map((k) => {
|
||||
const on = draft.base_features.includes(k);
|
||||
return (
|
||||
<label key={k} className="flex items-center gap-2.5 rounded border border-rule-soft bg-panel-2 px-2.5 py-2">
|
||||
<input type="checkbox" checked={on} onChange={(e) => toggleFeature(k, e.target.checked)} />
|
||||
<span>
|
||||
<span className="block text-[0.86rem]">{featureLabel(k)}</span>
|
||||
<span className="block text-[0.75rem] text-ink-3">{featureDesc(k)}</span>
|
||||
</span>
|
||||
<span className="ml-auto font-mono text-[0.66rem] uppercase tracking-[0.1em] text-ink-3">{on ? "Included" : "Sold as add-on"}</span>
|
||||
</label>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
<p className="text-[0.78rem] text-ink-3">No tier bundles a feature today. Including one here grants it with the plan and removes it from the customer's purchase form.</p>
|
||||
</section>
|
||||
|
||||
<section className="grid gap-2">
|
||||
<span className="font-mono text-[0.66rem] uppercase tracking-[0.14em] text-ink-3">Availability</span>
|
||||
<div className="grid gap-2 sm:grid-cols-2">
|
||||
<label className="grid gap-1">
|
||||
<span className="text-[0.78rem] text-ink-3">Support level</span>
|
||||
<select value={draft.support_level} onChange={(e) => setDraft({ ...draft, support_level: e.target.value })} className={controlClass("h-9 text-[0.84rem]")}>
|
||||
{SUPPORT_LEVELS.map((s) => (
|
||||
<option key={s.value} value={s.value}>
|
||||
{s.label}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
</label>
|
||||
<label className="flex items-center gap-2 self-end pb-2 text-[0.86rem]">
|
||||
<input type="checkbox" checked={draft.active} onChange={(e) => setDraft({ ...draft, active: e.target.checked })} />
|
||||
Offered to customers
|
||||
</label>
|
||||
</div>
|
||||
</section>
|
||||
</Modal>
|
||||
);
|
||||
}
|
||||
|
||||
/*
|
||||
* The plans half of /staff/pricing. It is a section rather than a page because
|
||||
* a tier's allowances and a tier's price are one decision made in one sitting,
|
||||
* and they were two screens with no view showing both.
|
||||
*/
|
||||
export function PlansSection() {
|
||||
const qc = useQueryClient();
|
||||
const plans = useQuery({ queryKey: ["plans"], queryFn: api.staff.plans });
|
||||
const licenses = useQuery({ queryKey: ["staff-licenses"], queryFn: () => api.staff.licenses() });
|
||||
|
||||
/* Two pieces of state, not one: `editing` is the plan whose modal is open,
|
||||
* `confirming` is the edit awaiting the change summary. Collapsing them put
|
||||
* the confirmation behind the modal it was confirming. */
|
||||
const [editing, setEditing] = useState<Plan | null>(null);
|
||||
const [confirming, setConfirming] = useState<Plan | null>(null);
|
||||
|
||||
const save = useMutation({
|
||||
mutationFn: (p: Plan) => api.staff.updatePlan(p.deployment, p.tier, p),
|
||||
onSuccess: () => {
|
||||
qc.invalidateQueries({ queryKey: ["plans"] });
|
||||
setConfirming(null);
|
||||
},
|
||||
});
|
||||
|
||||
const original = plans.data?.find((p) => p.deployment === confirming?.deployment && p.tier === confirming?.tier);
|
||||
|
||||
return (
|
||||
<div className="grid gap-6">
|
||||
<SectionHeading title="Plans" note="What each tier grants. Open a tier to change its base limits and features. Every issued licence snapshots the plan it was cut from, so editing one never rewrites an existing licence." />
|
||||
|
||||
{confirming && original && (
|
||||
<ConfirmPlanChange
|
||||
plan={original}
|
||||
next={confirming}
|
||||
issuedCount={(licenses.data ?? []).filter((l) => l.tier === confirming.tier && l.deployment === confirming.deployment).length}
|
||||
onConfirm={() => save.mutate(confirming)}
|
||||
onCancel={() => setConfirming(null)}
|
||||
/>
|
||||
)}
|
||||
|
||||
{(["cloud", "self_hosted"] as const).map((deployment: Deployment) => (
|
||||
<section key={deployment} className="grid gap-2.5">
|
||||
<h2 className="font-mono text-[0.68rem] uppercase tracking-[0.14em] text-ink-3">{deployment === "cloud" ? "Cloud" : "Self-hosted"}</h2>
|
||||
<div className="grid gap-2.5 sm:grid-cols-2 lg:grid-cols-3">
|
||||
{(plans.data ?? [])
|
||||
.filter((p) => p.deployment === deployment)
|
||||
.map((p) => (
|
||||
<TierCard key={planKey(p)} plan={p} onOpen={() => setEditing(p)} />
|
||||
))}
|
||||
</div>
|
||||
</section>
|
||||
))}
|
||||
|
||||
{editing && (
|
||||
<PlanModal
|
||||
key={planKey(editing)}
|
||||
plan={editing}
|
||||
onClose={() => setEditing(null)}
|
||||
onSave={(next) => {
|
||||
setEditing(null);
|
||||
setConfirming(next);
|
||||
}}
|
||||
/>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
/*
|
||||
* The heading that separates the two halves of /staff/pricing.
|
||||
*
|
||||
* It is not PageHeader: the page has one of those, and a second title-sized
|
||||
* heading under it would read as a second page. This is the same mono eyebrow
|
||||
* idiom the deployment groups use, one level up.
|
||||
*/
|
||||
export function SectionHeading({ title, note }: { title: string; note: string }) {
|
||||
return (
|
||||
<div className="grid gap-1 border-b border-rule pb-2">
|
||||
<h2 className="text-[1.05rem] font-bold tracking-[-0.01em]">{title}</h2>
|
||||
<p className="max-w-[68ch] text-[0.84rem] text-ink-3">{note}</p>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
"use client";
|
||||
|
||||
import { PageHeader } from "@/components/PageHeader";
|
||||
import { CatalogueSection } from "./CatalogueSection";
|
||||
import { PlansSection } from "./PlansSection";
|
||||
|
||||
/*
|
||||
* Plans and catalogue on one page.
|
||||
*
|
||||
* They were two nav entries, and the split asked staff to hold one half in
|
||||
* their head while looking at the other: a tier's allowances decide what the
|
||||
* metered component charges for, and the base fee is meaningless without the
|
||||
* allowance it includes. One page, two sections, in the order the decision is
|
||||
* made — what a tier grants, then what it costs.
|
||||
*/
|
||||
export default function PricingPage() {
|
||||
return (
|
||||
<div className="grid gap-7">
|
||||
<PageHeader title="Pricing" back={{ href: "/staff", label: "Operations" }} subtitle="What each tier grants, and what every priceable component costs." />
|
||||
<PlansSection />
|
||||
<CatalogueSection />
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
"use client";
|
||||
|
||||
import { useEffect, useRef } from "react";
|
||||
|
||||
/*
|
||||
* A native <dialog>, not a div with a fixed overlay.
|
||||
*
|
||||
* showModal() gives focus trapping, inert background, Escape and the top layer
|
||||
* for free — all four are things a hand-rolled overlay gets wrong, and the third
|
||||
* is the one staff will actually reach for. The only wiring needed is keeping
|
||||
* React state and the element's open state in step, and routing every close —
|
||||
* Escape, backdrop, button — through one onClose.
|
||||
*/
|
||||
export function Modal({
|
||||
open,
|
||||
onClose,
|
||||
title,
|
||||
meta,
|
||||
footer,
|
||||
children,
|
||||
}: {
|
||||
open: boolean;
|
||||
onClose: () => void;
|
||||
title: string;
|
||||
meta?: React.ReactNode;
|
||||
footer?: React.ReactNode;
|
||||
children: React.ReactNode;
|
||||
}) {
|
||||
const ref = useRef<HTMLDialogElement>(null);
|
||||
|
||||
useEffect(() => {
|
||||
const el = ref.current;
|
||||
if (!el) return;
|
||||
if (open && !el.open) el.showModal();
|
||||
if (!open && el.open) el.close();
|
||||
}, [open]);
|
||||
|
||||
return (
|
||||
<dialog
|
||||
ref={ref}
|
||||
onCancel={(e) => {
|
||||
e.preventDefault();
|
||||
onClose();
|
||||
}}
|
||||
/* Clicking the backdrop hits the dialog element itself, never a
|
||||
* child — so this closes on backdrop and not on content. */
|
||||
onClick={(e) => {
|
||||
if (e.target === ref.current) onClose();
|
||||
}}
|
||||
className="w-[min(44rem,94vw)] rounded border border-rule bg-panel p-0 text-ink shadow-lg backdrop:bg-[rgba(4,12,24,0.55)]"
|
||||
>
|
||||
<header className="flex flex-wrap items-center gap-3 border-b border-rule-soft bg-panel-2 px-4 py-3">
|
||||
<h2 className="text-[1.02rem] font-bold tracking-[-0.01em]">{title}</h2>
|
||||
{meta && <span className="font-mono text-[0.68rem] uppercase tracking-[0.12em] text-ink-3">{meta}</span>}
|
||||
<button type="button" onClick={onClose} className="ml-auto rounded border border-rule px-2 py-1 font-mono text-[0.68rem] uppercase tracking-[0.1em] text-ink-2 hover:border-ink-3" aria-label="Close">
|
||||
Esc
|
||||
</button>
|
||||
</header>
|
||||
<div className="grid max-h-[68vh] gap-4 overflow-y-auto p-4">{children}</div>
|
||||
{footer && <footer className="flex flex-wrap items-center gap-3 border-t border-rule-soft bg-panel-2 px-4 py-3">{footer}</footer>}
|
||||
</dialog>
|
||||
);
|
||||
}
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
import { useMemo } from "react";
|
||||
import type { CatalogueRow, Deployment, Plan, Term, Tier } from "@/lib/api";
|
||||
import { rowsForPlan } from "@/lib/catalogue";
|
||||
import { featureLabel } from "@/lib/features";
|
||||
|
||||
export interface PlanChoice {
|
||||
@@ -50,7 +51,7 @@ export default function PlanConfigurator({
|
||||
);
|
||||
const plan = available.find((p) => p.tier === value.tier);
|
||||
const rows = useMemo(
|
||||
() => catalogue.filter((r) => r.deployment === deployment && r.tier === value.tier),
|
||||
() => rowsForPlan(catalogue, deployment, value.tier),
|
||||
[catalogue, deployment, value.tier],
|
||||
);
|
||||
const featureRows = rows.filter((r) => r.kind === "feature");
|
||||
|
||||
@@ -8,6 +8,9 @@
|
||||
* customer-facing and should be shown verbatim).
|
||||
*/
|
||||
|
||||
/* catalogue.ts imports only types from here, so this is not a cycle. */
|
||||
import { rowsForPlan } from "@/lib/catalogue";
|
||||
|
||||
export const API_BASE = (process.env.NEXT_PUBLIC_ADMIN_API_URL ?? "").replace(/\/$/, "");
|
||||
|
||||
export class NotConnected extends Error {
|
||||
@@ -176,6 +179,11 @@ export interface Plan {
|
||||
|
||||
export interface CatalogueRow {
|
||||
kind: "base" | "limit" | "feature";
|
||||
/* "plan" rows carry a deployment and tier and belong to that plan alone.
|
||||
* "shared" rows leave both empty and are sold by every paid plan, which is
|
||||
* why a price ID is typed once rather than four times. Read them through
|
||||
* rowsForPlan in lib/catalogue, never by filtering on deployment. */
|
||||
scope: "plan" | "shared";
|
||||
deployment: Deployment;
|
||||
tier: Tier;
|
||||
limit_key?: string;
|
||||
@@ -213,9 +221,7 @@ export function lineItemsFor(
|
||||
const env = opts.env;
|
||||
const plan = opts.plans.find((p) => p.deployment === deployment && p.tier === choice.tier);
|
||||
if (!plan) return [];
|
||||
const rows = opts.catalogue.filter(
|
||||
(r) => r.deployment === deployment && r.tier === choice.tier,
|
||||
);
|
||||
const rows = rowsForPlan(opts.catalogue, deployment, choice.tier);
|
||||
const priceOf = (r: CatalogueRow) => r.price_ids?.[env]?.[choice.term] ?? "";
|
||||
const base = plan.base_limits.max_servers;
|
||||
const items: { priceId: string; quantity: number }[] = [];
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
import type { CatalogueRow, Deployment, Tier } from "@/lib/api";
|
||||
|
||||
/*
|
||||
* rowsForPlan is the TypeScript half of Go's models.CatalogueFor, and the two
|
||||
* must change together — the same shape of hazard as web/lib/targets.ts.
|
||||
*
|
||||
* A plan sells its own base row plus every shared add-on row. Shared rows leave
|
||||
* deployment and tier empty, so the filter this replaced — `r.deployment === dep
|
||||
* && r.tier === tier` — now returns a plan priced by its base fee and nothing
|
||||
* else. There were five copies of that filter; this is why it is a module.
|
||||
*/
|
||||
export function rowsForPlan(
|
||||
catalogue: CatalogueRow[],
|
||||
deployment: Deployment,
|
||||
tier: Tier,
|
||||
): CatalogueRow[] {
|
||||
return catalogue.filter(
|
||||
(r) => r.scope === "shared" || (r.deployment === deployment && r.tier === tier),
|
||||
);
|
||||
}
|
||||
|
||||
/* Every add-on a paid plan can be sold, in one list. The staff catalogue editor
|
||||
* shows these once; the purchase form reads them per plan through rowsForPlan. */
|
||||
export function sharedRows(catalogue: CatalogueRow[]): CatalogueRow[] {
|
||||
return catalogue.filter((r) => r.scope === "shared");
|
||||
}
|
||||
|
||||
/* The base fee rows, which are genuinely one per plan because each is its own
|
||||
* Paddle product at its own price. */
|
||||
export function planRows(catalogue: CatalogueRow[]): CatalogueRow[] {
|
||||
return catalogue.filter((r) => r.scope !== "shared");
|
||||
}
|
||||
|
||||
/* A stable identity for a row, used as a React key and as the draft key in the
|
||||
* staff editor. Mirrors the natural key the API addresses a row by. */
|
||||
export function rowKey(r: CatalogueRow): string {
|
||||
return [r.scope ?? "plan", r.deployment ?? "", r.tier ?? "", r.kind, r.limit_key ?? "", r.feature_key ?? ""].join("/");
|
||||
}
|
||||
@@ -10,12 +10,14 @@ export const FEATURE_LABEL: Record<string, string> = {
|
||||
console: "Browser console",
|
||||
oidc: "Single sign-on",
|
||||
vuln_scanning: "Vulnerability scanning",
|
||||
status_pages: "Status pages",
|
||||
};
|
||||
|
||||
export const FEATURE_DESC: Record<string, string> = {
|
||||
console: "In-browser SSH, RDP and VNC sessions",
|
||||
oidc: "OIDC sign-in for your whole team",
|
||||
vuln_scanning: "Package inventory matched against distribution security advisories",
|
||||
status_pages: "Public status pages for your customers, built from your monitors",
|
||||
};
|
||||
|
||||
export function featureLabel(key: string): string {
|
||||
|
||||
@@ -8,6 +8,14 @@ import type { NextConfig } from "next";
|
||||
*/
|
||||
const nextConfig: NextConfig = {
|
||||
output: "standalone",
|
||||
/* Plans and catalogue became one page. Both old paths are bookmarked in
|
||||
* staff browsers, so they redirect rather than 404. */
|
||||
async redirects() {
|
||||
return [
|
||||
{ source: "/staff/plans", destination: "/staff/pricing", permanent: true },
|
||||
{ source: "/staff/catalogue", destination: "/staff/pricing", permanent: true },
|
||||
];
|
||||
},
|
||||
};
|
||||
|
||||
export default nextConfig;
|
||||
|
||||
@@ -18,6 +18,10 @@ const (
|
||||
TypeTCP = "tcp"
|
||||
TypeICMP = "icmp"
|
||||
TypeTLS = "tls"
|
||||
|
||||
// UserAgent identifies Vantage monitor traffic so a WAF rule can single it
|
||||
// out. Match on a prefix, not equality: the version moves.
|
||||
UserAgent = "Vantage-Monitor/1.0 (+https://vantage.hostxtra.co.uk)"
|
||||
)
|
||||
|
||||
|
||||
@@ -84,6 +88,7 @@ func runHTTP(ctx context.Context, s Spec) Result {
|
||||
if err != nil {
|
||||
return Result{Message: err.Error()}
|
||||
}
|
||||
req.Header.Set("User-Agent", UserAgent)
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return Result{LatencyMs: msSince(start), Message: err.Error()}
|
||||
|
||||
@@ -1,5 +1,3 @@
|
||||
|
||||
|
||||
package pb
|
||||
|
||||
import (
|
||||
@@ -83,8 +81,6 @@ type UploadKeyResponse struct {
|
||||
KeyId string `json:"key_id"`
|
||||
}
|
||||
|
||||
|
||||
|
||||
type PackageUpdate struct {
|
||||
Name string `json:"name"`
|
||||
CurrentVersion string `json:"current_version,omitempty"`
|
||||
@@ -99,8 +95,6 @@ type ReportUpdatesRequest struct {
|
||||
|
||||
type ReportUpdatesResponse struct{}
|
||||
|
||||
|
||||
|
||||
type CPUReport struct {
|
||||
Model string `json:"model,omitempty"`
|
||||
Cores int `json:"cores,omitempty"`
|
||||
@@ -119,20 +113,19 @@ type PartitionReport struct {
|
||||
UsedBytes uint64 `json:"used_bytes"`
|
||||
}
|
||||
type InventoryReport struct {
|
||||
ServerId string `json:"server_id"`
|
||||
AgentToken string `json:"agent_token"`
|
||||
IncludeStatic bool `json:"include_static"`
|
||||
CPU *CPUReport `json:"cpu,omitempty"`
|
||||
Memory *MemReport `json:"memory,omitempty"`
|
||||
SwapTotal uint64 `json:"swap_total"`
|
||||
SwapUsed uint64 `json:"swap_used"`
|
||||
Partitions []PartitionReport `json:"partitions,omitempty"`
|
||||
Kernel string `json:"kernel,omitempty"`
|
||||
ServerId string `json:"server_id"`
|
||||
AgentToken string `json:"agent_token"`
|
||||
IncludeStatic bool `json:"include_static"`
|
||||
CPU *CPUReport `json:"cpu,omitempty"`
|
||||
Memory *MemReport `json:"memory,omitempty"`
|
||||
SwapTotal uint64 `json:"swap_total"`
|
||||
SwapUsed uint64 `json:"swap_used"`
|
||||
Partitions []PartitionReport `json:"partitions,omitempty"`
|
||||
Kernel string `json:"kernel,omitempty"`
|
||||
RebootRequired bool `json:"reboot_required,omitempty"`
|
||||
}
|
||||
type InventoryReportResponse struct{}
|
||||
|
||||
|
||||
|
||||
type MonitorSpec struct {
|
||||
MonitorId string `json:"monitor_id"`
|
||||
Type string `json:"type"`
|
||||
@@ -217,8 +210,6 @@ type ServerCommand struct {
|
||||
// keepalive is not sufficient on its own.
|
||||
type PingCmd struct{}
|
||||
|
||||
|
||||
|
||||
type CleanupWorkspaceCmd struct {
|
||||
WorkspaceId string `json:"workspace_id"`
|
||||
}
|
||||
@@ -241,12 +232,12 @@ type GenerateKeyCmd struct {
|
||||
}
|
||||
|
||||
type AgentMessage struct {
|
||||
ServerId string `json:"server_id"`
|
||||
AgentToken string `json:"agent_token"`
|
||||
Ready *AgentReady `json:"ready,omitempty"`
|
||||
Result *CommandResult `json:"result,omitempty"`
|
||||
StepResult *StepResult `json:"step_result,omitempty"`
|
||||
StepOutput *StepOutputChunk `json:"step_output,omitempty"`
|
||||
ServerId string `json:"server_id"`
|
||||
AgentToken string `json:"agent_token"`
|
||||
Ready *AgentReady `json:"ready,omitempty"`
|
||||
Result *CommandResult `json:"result,omitempty"`
|
||||
StepResult *StepResult `json:"step_result,omitempty"`
|
||||
StepOutput *StepOutputChunk `json:"step_output,omitempty"`
|
||||
|
||||
WorkloadLogsResult *WorkloadLogsResult `json:"workload_logs_result,omitempty"`
|
||||
}
|
||||
@@ -264,8 +255,7 @@ type RunStepCmd struct {
|
||||
Script string `json:"script"`
|
||||
Env map[string]string `json:"env,omitempty"`
|
||||
TimeoutSeconds int `json:"timeout_seconds,omitempty"`
|
||||
|
||||
|
||||
|
||||
WorkspaceId string `json:"workspace_id,omitempty"`
|
||||
}
|
||||
|
||||
@@ -284,8 +274,6 @@ type StepOutputChunk struct {
|
||||
Eof bool `json:"eof,omitempty"`
|
||||
}
|
||||
|
||||
|
||||
|
||||
type Vantage_CommandStreamClient interface {
|
||||
Send(*AgentMessage) error
|
||||
Recv() (*ServerCommand, error)
|
||||
@@ -308,8 +296,6 @@ func (c *vantageCommandStreamClient) Recv() (*ServerCommand, error) {
|
||||
return m, nil
|
||||
}
|
||||
|
||||
|
||||
|
||||
type Vantage_CommandStreamServer interface {
|
||||
Send(*ServerCommand) error
|
||||
Recv() (*AgentMessage, error)
|
||||
|
||||
@@ -450,6 +450,16 @@ func runInventory(ctx context.Context, cfg *config.Config) {
|
||||
r := inventory.Collect(static)
|
||||
r.ServerId = cfg.ServerID
|
||||
r.AgentToken = cfg.AgentToken
|
||||
// Static snapshots only — every 15 minutes, not every 30 seconds. On
|
||||
// Windows this spawns a PowerShell process, which is not something to
|
||||
// do twice a minute forever, and a host rebooted by hand clearing the
|
||||
// flag within a quarter of an hour is soon enough.
|
||||
//
|
||||
// Computed here rather than inside inventory.Collect so the inventory
|
||||
// package gains no dependency on updates.
|
||||
if static {
|
||||
r.RebootRequired = updates.RebootRequired()
|
||||
}
|
||||
if err := client.ReportInventory(r); err != nil {
|
||||
log.Printf("report inventory: %v", err)
|
||||
}
|
||||
|
||||
@@ -3,7 +3,6 @@ package agentsync
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"runtime"
|
||||
"time"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/agent/internal/config"
|
||||
@@ -18,10 +17,6 @@ const workloadInterval = 60 * time.Second
|
||||
|
||||
// runWorkloads reports what this host runs, on its own ticker.
|
||||
func runWorkloads(ctx context.Context, cfg *config.Config) {
|
||||
if runtime.GOOS != "linux" {
|
||||
return
|
||||
}
|
||||
|
||||
reportWorkloads(cfg)
|
||||
|
||||
ticker := time.NewTicker(workloadInterval)
|
||||
@@ -42,10 +37,6 @@ func runWorkloads(ctx context.Context, cfg *config.Config) {
|
||||
// This is the ONLY writer of the server_workloads collection. RefreshWorkloadsCmd
|
||||
// calls straight into here rather than answering with data of its own.
|
||||
func reportWorkloads(cfg *config.Config) {
|
||||
if runtime.GOOS != "linux" {
|
||||
return
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||
defer cancel()
|
||||
|
||||
|
||||
@@ -1,238 +1,24 @@
|
||||
package updates
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// PackageUpdate is one pending update. On Linux it is a package with a version
|
||||
// on each side. On Windows CurrentVersion is empty and NewVersion carries the
|
||||
// KB article ID: a Windows update is not a version bump of a named package,
|
||||
// and inventing a current version would put a wrong string in front of an
|
||||
// operator.
|
||||
type PackageUpdate struct {
|
||||
Name string
|
||||
CurrentVersion string
|
||||
NewVersion string
|
||||
}
|
||||
|
||||
func detectPM() string {
|
||||
for _, pm := range []string{"apt-get", "dnf", "yum", "pacman", "zypper", "apk"} {
|
||||
if _, err := exec.LookPath(pm); err == nil {
|
||||
if pm == "apt-get" {
|
||||
return "apt"
|
||||
}
|
||||
return pm
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
// CheckAvailable lists pending OS updates.
|
||||
func CheckAvailable() ([]PackageUpdate, error) { return checkAvailable() }
|
||||
|
||||
// ApplyAll installs every pending update. It never reboots: a control plane
|
||||
// silently restarting a production server is unrecoverable from the UI, so the
|
||||
// reboot stays a decision a person or a workflow makes. RebootRequired reports
|
||||
// when one is owed.
|
||||
func ApplyAll() error { return applyAll() }
|
||||
|
||||
|
||||
func CheckAvailable() ([]PackageUpdate, error) {
|
||||
switch detectPM() {
|
||||
case "apt":
|
||||
return checkApt()
|
||||
case "dnf":
|
||||
return checkDnfYum("dnf")
|
||||
case "yum":
|
||||
return checkDnfYum("yum")
|
||||
case "pacman":
|
||||
return checkPacman()
|
||||
case "zypper":
|
||||
return checkZypper()
|
||||
case "apk":
|
||||
return checkApk()
|
||||
default:
|
||||
return nil, nil
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
func ApplyAll() error {
|
||||
switch detectPM() {
|
||||
case "apt":
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||
defer cancel()
|
||||
if err := exec.CommandContext(ctx, "apt-get", "update", "-qq").Run(); err != nil {
|
||||
return err
|
||||
}
|
||||
return exec.CommandContext(ctx, "apt-get", "upgrade", "-y").Run()
|
||||
case "dnf":
|
||||
return exec.Command("dnf", "upgrade", "-y").Run()
|
||||
case "yum":
|
||||
return exec.Command("yum", "upgrade", "-y").Run()
|
||||
case "pacman":
|
||||
return exec.Command("pacman", "-Syu", "--noconfirm").Run()
|
||||
case "zypper":
|
||||
return exec.Command("zypper", "update", "-y").Run()
|
||||
case "apk":
|
||||
return exec.Command("apk", "upgrade").Run()
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
func checkApt() ([]PackageUpdate, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||
defer cancel()
|
||||
|
||||
exec.CommandContext(ctx, "apt-get", "update", "-qq").Run()
|
||||
|
||||
out, err := exec.Command("apt", "list", "--upgradable").Output()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var updates []PackageUpdate
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
|
||||
if !strings.Contains(line, "[upgradable from:") {
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(line)
|
||||
if len(parts) < 2 {
|
||||
continue
|
||||
}
|
||||
name := strings.SplitN(parts[0], "/", 2)[0]
|
||||
newVer := parts[1]
|
||||
oldVer := ""
|
||||
if idx := strings.Index(line, "upgradable from: "); idx != -1 {
|
||||
rest := line[idx+len("upgradable from: "):]
|
||||
oldVer = strings.TrimSuffix(strings.TrimSpace(rest), "]")
|
||||
}
|
||||
updates = append(updates, PackageUpdate{Name: name, CurrentVersion: oldVer, NewVersion: newVer})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func checkDnfYum(pm string) ([]PackageUpdate, error) {
|
||||
cmd := exec.Command(pm, "check-update")
|
||||
out, err := cmd.Output()
|
||||
if exitErr, ok := err.(*exec.ExitError); ok && exitErr.ExitCode() == 100 {
|
||||
err = nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var updates []PackageUpdate
|
||||
pastHeader := false
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
if !pastHeader {
|
||||
if strings.TrimSpace(line) == "" {
|
||||
pastHeader = true
|
||||
}
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(line)
|
||||
if len(parts) < 2 {
|
||||
continue
|
||||
}
|
||||
|
||||
name := strings.SplitN(parts[0], ".", 2)[0]
|
||||
updates = append(updates, PackageUpdate{Name: name, NewVersion: parts[1]})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func checkPacman() ([]PackageUpdate, error) {
|
||||
out, _ := exec.Command("pacman", "-Qu").Output()
|
||||
var updates []PackageUpdate
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
parts := strings.Fields(scanner.Text())
|
||||
|
||||
if len(parts) < 4 {
|
||||
continue
|
||||
}
|
||||
updates = append(updates, PackageUpdate{Name: parts[0], CurrentVersion: parts[1], NewVersion: parts[3]})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func checkZypper() ([]PackageUpdate, error) {
|
||||
out, err := exec.Command("zypper", "list-updates").Output()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var updates []PackageUpdate
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
|
||||
if !strings.HasPrefix(line, "v |") && !strings.HasPrefix(line, "i |") {
|
||||
continue
|
||||
}
|
||||
parts := strings.Split(line, "|")
|
||||
if len(parts) < 5 {
|
||||
continue
|
||||
}
|
||||
updates = append(updates, PackageUpdate{
|
||||
Name: strings.TrimSpace(parts[2]),
|
||||
CurrentVersion: strings.TrimSpace(parts[3]),
|
||||
NewVersion: strings.TrimSpace(parts[4]),
|
||||
})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func checkApk() ([]PackageUpdate, error) {
|
||||
out, err := exec.Command("apk", "list", "--upgradable").Output()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var updates []PackageUpdate
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
if !strings.Contains(line, "[upgradable") {
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(line)
|
||||
if len(parts) < 1 {
|
||||
continue
|
||||
}
|
||||
pkgVer := parts[0]
|
||||
name := apkName(pkgVer)
|
||||
newVer := apkVersion(pkgVer)
|
||||
oldVer := ""
|
||||
if idx := strings.Index(line, "upgradable from:"); idx != -1 {
|
||||
rest := strings.TrimSpace(line[idx+len("upgradable from:"):])
|
||||
rest = strings.TrimSuffix(rest, "]")
|
||||
oldVer = apkVersion(strings.TrimSpace(rest))
|
||||
}
|
||||
updates = append(updates, PackageUpdate{Name: name, CurrentVersion: oldVer, NewVersion: newVer})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func apkName(pkgVer string) string {
|
||||
parts := strings.Split(pkgVer, "-")
|
||||
var name []string
|
||||
for _, p := range parts {
|
||||
if len(p) > 0 && p[0] >= '0' && p[0] <= '9' {
|
||||
break
|
||||
}
|
||||
name = append(name, p)
|
||||
}
|
||||
return strings.Join(name, "-")
|
||||
}
|
||||
|
||||
func apkVersion(pkgVer string) string {
|
||||
parts := strings.Split(pkgVer, "-")
|
||||
var ver []string
|
||||
inVer := false
|
||||
for _, p := range parts {
|
||||
if !inVer && len(p) > 0 && p[0] >= '0' && p[0] <= '9' {
|
||||
inVer = true
|
||||
}
|
||||
if inVer {
|
||||
ver = append(ver, p)
|
||||
}
|
||||
}
|
||||
return strings.Join(ver, "-")
|
||||
}
|
||||
// RebootRequired reports whether this host is waiting on a restart.
|
||||
func RebootRequired() bool { return rebootRequired() }
|
||||
|
||||
@@ -0,0 +1,252 @@
|
||||
package updates
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"os"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
func detectPM() string {
|
||||
for _, pm := range []string{"apt-get", "dnf", "yum", "pacman", "zypper", "apk"} {
|
||||
if _, err := exec.LookPath(pm); err == nil {
|
||||
if pm == "apt-get" {
|
||||
return "apt"
|
||||
}
|
||||
return pm
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
|
||||
|
||||
func checkAvailable() ([]PackageUpdate, error) {
|
||||
switch detectPM() {
|
||||
case "apt":
|
||||
return checkApt()
|
||||
case "dnf":
|
||||
return checkDnfYum("dnf")
|
||||
case "yum":
|
||||
return checkDnfYum("yum")
|
||||
case "pacman":
|
||||
return checkPacman()
|
||||
case "zypper":
|
||||
return checkZypper()
|
||||
case "apk":
|
||||
return checkApk()
|
||||
default:
|
||||
return nil, nil
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
func applyAll() error {
|
||||
switch detectPM() {
|
||||
case "apt":
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute)
|
||||
defer cancel()
|
||||
if err := exec.CommandContext(ctx, "apt-get", "update", "-qq").Run(); err != nil {
|
||||
return err
|
||||
}
|
||||
return exec.CommandContext(ctx, "apt-get", "upgrade", "-y").Run()
|
||||
case "dnf":
|
||||
return exec.Command("dnf", "upgrade", "-y").Run()
|
||||
case "yum":
|
||||
return exec.Command("yum", "upgrade", "-y").Run()
|
||||
case "pacman":
|
||||
return exec.Command("pacman", "-Syu", "--noconfirm").Run()
|
||||
case "zypper":
|
||||
return exec.Command("zypper", "update", "-y").Run()
|
||||
case "apk":
|
||||
return exec.Command("apk", "upgrade").Run()
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// rebootRequired reads what the distributions themselves record. Debian and
|
||||
// Ubuntu drop a file; the RPM family answers through needs-restarting, whose
|
||||
// exit code is 1 when a reboot is owed and 0 when it is not.
|
||||
func rebootRequired() bool {
|
||||
if _, err := os.Stat("/var/run/reboot-required"); err == nil {
|
||||
return true
|
||||
}
|
||||
if _, err := exec.LookPath("dnf"); err == nil {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
if err := exec.CommandContext(ctx, "dnf", "needs-restarting", "-r").Run(); err != nil {
|
||||
if ee, ok := err.(*exec.ExitError); ok && ee.ExitCode() == 1 {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func checkApt() ([]PackageUpdate, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||
defer cancel()
|
||||
|
||||
exec.CommandContext(ctx, "apt-get", "update", "-qq").Run()
|
||||
|
||||
out, err := exec.Command("apt", "list", "--upgradable").Output()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var updates []PackageUpdate
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
|
||||
if !strings.Contains(line, "[upgradable from:") {
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(line)
|
||||
if len(parts) < 2 {
|
||||
continue
|
||||
}
|
||||
name := strings.SplitN(parts[0], "/", 2)[0]
|
||||
newVer := parts[1]
|
||||
oldVer := ""
|
||||
if idx := strings.Index(line, "upgradable from: "); idx != -1 {
|
||||
rest := line[idx+len("upgradable from: "):]
|
||||
oldVer = strings.TrimSuffix(strings.TrimSpace(rest), "]")
|
||||
}
|
||||
updates = append(updates, PackageUpdate{Name: name, CurrentVersion: oldVer, NewVersion: newVer})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func checkDnfYum(pm string) ([]PackageUpdate, error) {
|
||||
cmd := exec.Command(pm, "check-update")
|
||||
out, err := cmd.Output()
|
||||
if exitErr, ok := err.(*exec.ExitError); ok && exitErr.ExitCode() == 100 {
|
||||
err = nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var updates []PackageUpdate
|
||||
pastHeader := false
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
if !pastHeader {
|
||||
if strings.TrimSpace(line) == "" {
|
||||
pastHeader = true
|
||||
}
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(line)
|
||||
if len(parts) < 2 {
|
||||
continue
|
||||
}
|
||||
|
||||
name := strings.SplitN(parts[0], ".", 2)[0]
|
||||
updates = append(updates, PackageUpdate{Name: name, NewVersion: parts[1]})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func checkPacman() ([]PackageUpdate, error) {
|
||||
out, _ := exec.Command("pacman", "-Qu").Output()
|
||||
var updates []PackageUpdate
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
parts := strings.Fields(scanner.Text())
|
||||
|
||||
if len(parts) < 4 {
|
||||
continue
|
||||
}
|
||||
updates = append(updates, PackageUpdate{Name: parts[0], CurrentVersion: parts[1], NewVersion: parts[3]})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func checkZypper() ([]PackageUpdate, error) {
|
||||
out, err := exec.Command("zypper", "list-updates").Output()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var updates []PackageUpdate
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
|
||||
if !strings.HasPrefix(line, "v |") && !strings.HasPrefix(line, "i |") {
|
||||
continue
|
||||
}
|
||||
parts := strings.Split(line, "|")
|
||||
if len(parts) < 5 {
|
||||
continue
|
||||
}
|
||||
updates = append(updates, PackageUpdate{
|
||||
Name: strings.TrimSpace(parts[2]),
|
||||
CurrentVersion: strings.TrimSpace(parts[3]),
|
||||
NewVersion: strings.TrimSpace(parts[4]),
|
||||
})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func checkApk() ([]PackageUpdate, error) {
|
||||
out, err := exec.Command("apk", "list", "--upgradable").Output()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var updates []PackageUpdate
|
||||
scanner := bufio.NewScanner(bytes.NewReader(out))
|
||||
for scanner.Scan() {
|
||||
line := scanner.Text()
|
||||
if !strings.Contains(line, "[upgradable") {
|
||||
continue
|
||||
}
|
||||
parts := strings.Fields(line)
|
||||
if len(parts) < 1 {
|
||||
continue
|
||||
}
|
||||
pkgVer := parts[0]
|
||||
name := apkName(pkgVer)
|
||||
newVer := apkVersion(pkgVer)
|
||||
oldVer := ""
|
||||
if idx := strings.Index(line, "upgradable from:"); idx != -1 {
|
||||
rest := strings.TrimSpace(line[idx+len("upgradable from:"):])
|
||||
rest = strings.TrimSuffix(rest, "]")
|
||||
oldVer = apkVersion(strings.TrimSpace(rest))
|
||||
}
|
||||
updates = append(updates, PackageUpdate{Name: name, CurrentVersion: oldVer, NewVersion: newVer})
|
||||
}
|
||||
return updates, nil
|
||||
}
|
||||
|
||||
func apkName(pkgVer string) string {
|
||||
parts := strings.Split(pkgVer, "-")
|
||||
var name []string
|
||||
for _, p := range parts {
|
||||
if len(p) > 0 && p[0] >= '0' && p[0] <= '9' {
|
||||
break
|
||||
}
|
||||
name = append(name, p)
|
||||
}
|
||||
return strings.Join(name, "-")
|
||||
}
|
||||
|
||||
func apkVersion(pkgVer string) string {
|
||||
parts := strings.Split(pkgVer, "-")
|
||||
var ver []string
|
||||
inVer := false
|
||||
for _, p := range parts {
|
||||
if !inVer && len(p) > 0 && p[0] >= '0' && p[0] <= '9' {
|
||||
inVer = true
|
||||
}
|
||||
if inVer {
|
||||
ver = append(ver, p)
|
||||
}
|
||||
}
|
||||
return strings.Join(ver, "-")
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
//go:build !linux && !windows
|
||||
|
||||
// The build constraint above is load-bearing: "_other" is not a GOOS suffix, so
|
||||
// without it this file compiles on Linux too and collides with updates_linux.go.
|
||||
package updates
|
||||
|
||||
func checkAvailable() ([]PackageUpdate, error) { return nil, nil }
|
||||
func applyAll() error { return nil }
|
||||
func rebootRequired() bool { return false }
|
||||
@@ -0,0 +1,117 @@
|
||||
package updates
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/agent/internal/winexec"
|
||||
)
|
||||
|
||||
const (
|
||||
// The first search after a boot contacts Microsoft Update (or WSUS) and is
|
||||
// routinely slow. Ten minutes is not generous, it is realistic.
|
||||
searchTimeout = 10 * time.Minute
|
||||
|
||||
// A patch-Tuesday cumulative genuinely takes this long to download and
|
||||
// install on a modest server.
|
||||
applyTimeout = 60 * time.Minute
|
||||
|
||||
rebootTimeout = 2 * time.Minute
|
||||
)
|
||||
|
||||
// The Windows Update COM API is used rather than the PSWindowsUpdate module: it
|
||||
// is present on every supported Windows, needs no PowerShell Gallery install,
|
||||
// and works unchanged against a WSUS server on an air-gapped fleet. The agent
|
||||
// runs as LocalSystem, which holds the rights it requires.
|
||||
const searchScript = `
|
||||
$ErrorActionPreference = 'Stop'
|
||||
$searcher = (New-Object -ComObject Microsoft.Update.Session).CreateUpdateSearcher()
|
||||
$result = $searcher.Search("IsInstalled=0 and Type='Software' and IsHidden=0")
|
||||
$rows = @()
|
||||
foreach ($u in $result.Updates) {
|
||||
$ids = @($u.KBArticleIDs)
|
||||
$kb = ''
|
||||
if ($ids.Count -gt 0) { $kb = [string]$ids[0] }
|
||||
$rows += [pscustomobject]@{ title = [string]$u.Title; kb = $kb }
|
||||
}
|
||||
ConvertTo-Json -InputObject @($rows) -Depth 3 -Compress
|
||||
`
|
||||
|
||||
const applyScript = `
|
||||
$ErrorActionPreference = 'Stop'
|
||||
$session = New-Object -ComObject Microsoft.Update.Session
|
||||
$result = $session.CreateUpdateSearcher().Search("IsInstalled=0 and Type='Software' and IsHidden=0")
|
||||
|
||||
$batch = New-Object -ComObject Microsoft.Update.UpdateColl
|
||||
foreach ($u in $result.Updates) {
|
||||
if ($u.InstallationBehavior.CanRequestUserInput) { continue }
|
||||
if (-not $u.EulaAccepted) {
|
||||
try { $u.AcceptEula() } catch { continue }
|
||||
}
|
||||
$null = $batch.Add($u)
|
||||
}
|
||||
|
||||
if ($batch.Count -eq 0) { Write-Output 'nothing-to-install'; exit 0 }
|
||||
|
||||
$downloader = $session.CreateUpdateDownloader()
|
||||
$downloader.Updates = $batch
|
||||
$null = $downloader.Download()
|
||||
|
||||
$installer = $session.CreateUpdateInstaller()
|
||||
$installer.Updates = $batch
|
||||
$r = $installer.Install()
|
||||
|
||||
Write-Output ('resultcode=' + $r.ResultCode)
|
||||
# 2 = succeeded, 3 = succeeded with errors. Anything else failed, and this
|
||||
# process must exit non-zero so the agent logs a failure rather than an ack.
|
||||
if ($r.ResultCode -ne 2 -and $r.ResultCode -ne 3) { exit 1 }
|
||||
exit 0
|
||||
`
|
||||
|
||||
const rebootScript = `
|
||||
$ErrorActionPreference = 'SilentlyContinue'
|
||||
$si = New-Object -ComObject Microsoft.Update.SystemInfo
|
||||
if ($si.RebootRequired) { Write-Output 'true'; exit 0 }
|
||||
$keys = @(
|
||||
'HKLM:\SOFTWARE\Microsoft\Windows\CurrentVersion\Component Based Servicing\RebootPending',
|
||||
'HKLM:\SOFTWARE\Microsoft\Windows\CurrentVersion\WindowsUpdate\Auto Update\RebootRequired'
|
||||
)
|
||||
foreach ($k in $keys) { if (Test-Path $k) { Write-Output 'true'; exit 0 } }
|
||||
$sm = Get-ItemProperty 'HKLM:\SYSTEM\CurrentControlSet\Control\Session Manager' -Name PendingFileRenameOperations
|
||||
if ($sm -and $sm.PendingFileRenameOperations) { Write-Output 'true'; exit 0 }
|
||||
Write-Output 'false'
|
||||
`
|
||||
|
||||
func checkAvailable() ([]PackageUpdate, error) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), searchTimeout)
|
||||
defer cancel()
|
||||
|
||||
out, err := winexec.Run(ctx, searchScript)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("windows update search: %w", err)
|
||||
}
|
||||
return parseUpdateSearch(out)
|
||||
}
|
||||
|
||||
func applyAll() error {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), applyTimeout)
|
||||
defer cancel()
|
||||
|
||||
if _, err := winexec.Run(ctx, applyScript); err != nil {
|
||||
return fmt.Errorf("windows update install: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func rebootRequired() bool {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), rebootTimeout)
|
||||
defer cancel()
|
||||
|
||||
out, err := winexec.Run(ctx, rebootScript)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
return strings.TrimSpace(out) == "true"
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
package updates
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// winUpdate is one row of the Windows Update searcher's output, in the shape
|
||||
// searchScript emits it.
|
||||
type winUpdate struct {
|
||||
Title string `json:"title"`
|
||||
KB string `json:"kb"`
|
||||
}
|
||||
|
||||
// parseUpdateSearch reads the searcher's JSON.
|
||||
//
|
||||
// It carries no build tag on purpose: this is the half of the Windows update
|
||||
// path that can be tested on a development machine, and the agent module has no
|
||||
// Windows CI.
|
||||
func parseUpdateSearch(jsonText string) ([]PackageUpdate, error) {
|
||||
s := strings.TrimSpace(jsonText)
|
||||
if s == "" || s == "null" {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
var rows []winUpdate
|
||||
if err := json.Unmarshal([]byte(s), &rows); err != nil {
|
||||
// ConvertTo-Json renders a one-element array as a bare object.
|
||||
var one winUpdate
|
||||
if err2 := json.Unmarshal([]byte(s), &one); err2 != nil {
|
||||
return nil, err
|
||||
}
|
||||
rows = []winUpdate{one}
|
||||
}
|
||||
|
||||
out := make([]PackageUpdate, 0, len(rows))
|
||||
for _, r := range rows {
|
||||
u := PackageUpdate{Name: r.Title}
|
||||
if kb := strings.TrimSpace(r.KB); kb != "" {
|
||||
if !strings.HasPrefix(strings.ToUpper(kb), "KB") {
|
||||
kb = "KB" + kb
|
||||
}
|
||||
u.NewVersion = kb
|
||||
}
|
||||
out = append(out, u)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
package updates
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestParseUpdateSearchArray(t *testing.T) {
|
||||
in := `[{"title":"2026-08 Cumulative Update for Windows Server 2022","kb":"5034123"},
|
||||
{"title":"Windows Malicious Software Removal Tool","kb":"890830"}]`
|
||||
|
||||
got, err := parseUpdateSearch(in)
|
||||
if err != nil {
|
||||
t.Fatalf("parseUpdateSearch: %v", err)
|
||||
}
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("got %d updates, want 2", len(got))
|
||||
}
|
||||
if got[0].Name != "2026-08 Cumulative Update for Windows Server 2022" {
|
||||
t.Errorf("Name = %q", got[0].Name)
|
||||
}
|
||||
if got[0].NewVersion != "KB5034123" {
|
||||
t.Errorf("NewVersion = %q, want KB5034123", got[0].NewVersion)
|
||||
}
|
||||
if got[0].CurrentVersion != "" {
|
||||
t.Errorf("CurrentVersion = %q, want empty", got[0].CurrentVersion)
|
||||
}
|
||||
}
|
||||
|
||||
// PowerShell 5.1's ConvertTo-Json collapses a one-element array into a bare
|
||||
// object. A host with exactly one pending update is common, and a parser that
|
||||
// only accepts arrays reports it as zero.
|
||||
func TestParseUpdateSearchSingleObject(t *testing.T) {
|
||||
got, err := parseUpdateSearch(`{"title":"Security Intelligence Update","kb":"2267602"}`)
|
||||
if err != nil {
|
||||
t.Fatalf("parseUpdateSearch: %v", err)
|
||||
}
|
||||
if len(got) != 1 || got[0].NewVersion != "KB2267602" {
|
||||
t.Fatalf("got %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseUpdateSearchNoKB(t *testing.T) {
|
||||
got, err := parseUpdateSearch(`[{"title":"Driver update for Contoso NIC","kb":""}]`)
|
||||
if err != nil {
|
||||
t.Fatalf("parseUpdateSearch: %v", err)
|
||||
}
|
||||
if len(got) != 1 || got[0].NewVersion != "" {
|
||||
t.Fatalf("got %+v, want one update with an empty NewVersion", got)
|
||||
}
|
||||
}
|
||||
|
||||
// An empty result set is "nothing pending", not a parse failure.
|
||||
func TestParseUpdateSearchEmpty(t *testing.T) {
|
||||
for _, in := range []string{"", " \r\n", "[]", "null"} {
|
||||
got, err := parseUpdateSearch(in)
|
||||
if err != nil {
|
||||
t.Fatalf("parseUpdateSearch(%q): %v", in, err)
|
||||
}
|
||||
if len(got) != 0 {
|
||||
t.Fatalf("parseUpdateSearch(%q) = %+v, want none", in, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A KB already carrying its prefix must not become KBKB5034123.
|
||||
func TestParseUpdateSearchPrefixedKB(t *testing.T) {
|
||||
got, _ := parseUpdateSearch(`[{"title":"x","kb":"KB5034123"}]`)
|
||||
if got[0].NewVersion != "KB5034123" {
|
||||
t.Fatalf("NewVersion = %q", got[0].NewVersion)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
// Package winexec runs PowerShell on Windows hosts.
|
||||
//
|
||||
// It exists because three subsystems — updates, workload collection and
|
||||
// workload logs — all need the same invocation, and because getting a
|
||||
// multi-line script past Go quoting, cmd.exe quoting and PowerShell's own
|
||||
// parser is a problem worth solving once.
|
||||
package winexec
|
||||
|
||||
import (
|
||||
"encoding/base64"
|
||||
"unicode/utf16"
|
||||
)
|
||||
|
||||
// EncodeCommand renders a script for powershell.exe -EncodedCommand: UTF-16LE,
|
||||
// no byte-order mark, base64. This is deliberately free of build tags so it is
|
||||
// tested on a Linux development machine like every other pure function here.
|
||||
func EncodeCommand(script string) string {
|
||||
units := utf16.Encode([]rune(script))
|
||||
b := make([]byte, 0, len(units)*2)
|
||||
for _, u := range units {
|
||||
b = append(b, byte(u), byte(u>>8))
|
||||
}
|
||||
return base64.StdEncoding.EncodeToString(b)
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
package winexec
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestEncodeCommand(t *testing.T) {
|
||||
// "hi" as UTF-16LE is 68 00 69 00, which base64-encodes to aABpAA==.
|
||||
if got := EncodeCommand("hi"); got != "aABpAA==" {
|
||||
t.Fatalf("EncodeCommand(hi) = %q, want aABpAA==", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEncodeCommandMultiline(t *testing.T) {
|
||||
// Only that it round-trips through the same encoding PowerShell expects:
|
||||
// every ASCII byte followed by a zero byte, no BOM.
|
||||
got := EncodeCommand("a\nb")
|
||||
want := "YQAKAGIA"
|
||||
if got != want {
|
||||
t.Fatalf("EncodeCommand = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package winexec
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Run executes a PowerShell script and returns its stdout.
|
||||
//
|
||||
// powershell.exe rather than pwsh: everything this agent runs through here
|
||||
// touches Windows Update COM or CIM, both of which are most reliable under
|
||||
// Windows PowerShell 5.1, and 5.1 is present on every supported Windows while
|
||||
// pwsh is an optional install.
|
||||
func Run(ctx context.Context, script string) (string, error) {
|
||||
cmd := exec.CommandContext(ctx, "powershell.exe",
|
||||
"-NoProfile", "-NonInteractive", "-EncodedCommand", EncodeCommand(script))
|
||||
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
// Checked before the ExitError/stderr branch: CommandContext kills the
|
||||
// process on timeout, and that kill can itself produce an ExitError
|
||||
// carrying stderr text, so a genuine timeout would otherwise surface
|
||||
// as that stderr instead of the "timed out" message callers match on.
|
||||
if ctx.Err() == context.DeadlineExceeded {
|
||||
return "", fmt.Errorf("powershell: timed out")
|
||||
}
|
||||
if ee, ok := err.(*exec.ExitError); ok && len(ee.Stderr) > 0 {
|
||||
return "", fmt.Errorf("powershell: %s", strings.TrimSpace(string(ee.Stderr)))
|
||||
}
|
||||
return "", fmt.Errorf("powershell: %w", err)
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
@@ -4,9 +4,6 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
@@ -14,34 +11,12 @@ import (
|
||||
// ErrProtected is returned for a workload the agent will not act on.
|
||||
var ErrProtected = errors.New("workload is protected")
|
||||
|
||||
// AgentUnit is the systemd unit this agent runs as.
|
||||
const AgentUnit = "vantage-agent.service"
|
||||
|
||||
// controlTimeout bounds a stop that may never finish on its own. `docker stop`
|
||||
// waits on a container that may ignore SIGTERM, and `systemctl stop` on a unit
|
||||
// with a long TimeoutStopSec blocks for exactly as long as that says. A
|
||||
// waits on a container that may ignore SIGTERM, and both systemctl and
|
||||
// Stop-Service block for as long as the unit's own stop timeout says. A
|
||||
// timeout must return a real error rather than an ack implying success.
|
||||
const controlTimeout = 90 * time.Second
|
||||
|
||||
// ownContainerID is read once: the container this agent runs in, if any.
|
||||
var ownContainerID = detectOwnContainer()
|
||||
|
||||
var cgroupContainerRe = regexp.MustCompile(`[0-9a-f]{64}`)
|
||||
|
||||
// detectOwnContainer returns this process's container ID, or "" on a host
|
||||
// install. The agent is normally a systemd service, so "" is the common case;
|
||||
// this exists so containerising it later cannot silently remove the guard.
|
||||
func detectOwnContainer() string {
|
||||
b, err := os.ReadFile("/proc/self/cgroup")
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
if m := cgroupContainerRe.FindString(string(b)); m != "" {
|
||||
return m
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// isProtected reports whether the agent refuses to act on this workload.
|
||||
//
|
||||
// The refusal lives here, in the agent, and not in the control plane. As with
|
||||
@@ -50,7 +25,7 @@ func detectOwnContainer() string {
|
||||
// denylist alone would be bypassed by the next dispatch path someone adds.
|
||||
func isProtected(kind, id, name string) bool {
|
||||
if kind == "unit" {
|
||||
return id == AgentUnit || name == strings.TrimSuffix(AgentUnit, ".service")
|
||||
return isProtectedUnit(id, name)
|
||||
}
|
||||
if ownContainerID == "" {
|
||||
return false
|
||||
@@ -85,21 +60,5 @@ func Control(ctx context.Context, kind, id, action string) error {
|
||||
ctx, cancel := context.WithTimeout(ctx, controlTimeout)
|
||||
defer cancel()
|
||||
|
||||
var cmd *exec.Cmd
|
||||
switch kind {
|
||||
case "container":
|
||||
cmd = exec.CommandContext(ctx, "docker", action, id)
|
||||
case "unit":
|
||||
cmd = exec.CommandContext(ctx, "systemctl", action, id)
|
||||
default:
|
||||
return fmt.Errorf("unknown workload kind %q", kind)
|
||||
}
|
||||
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
if ctx.Err() == context.DeadlineExceeded {
|
||||
return fmt.Errorf("%s %s timed out after %s", action, id, controlTimeout)
|
||||
}
|
||||
return fmt.Errorf("%s %s: %s", action, id, strings.TrimSpace(string(out)))
|
||||
}
|
||||
return nil
|
||||
return controlPlatform(ctx, kind, id, action)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
package workloads
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"regexp"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// AgentUnit is the systemd unit this agent runs as.
|
||||
const AgentUnit = "vantage-agent.service"
|
||||
|
||||
// ownContainerID is read once: the container this agent runs in, if any.
|
||||
var ownContainerID = detectOwnContainer()
|
||||
|
||||
var cgroupContainerRe = regexp.MustCompile(`[0-9a-f]{64}`)
|
||||
|
||||
// detectOwnContainer returns this process's container ID, or "" on a host
|
||||
// install. The agent is normally a systemd service, so "" is the common case;
|
||||
// this exists so containerising it later cannot silently remove the guard.
|
||||
func detectOwnContainer() string {
|
||||
b, err := os.ReadFile("/proc/self/cgroup")
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
if m := cgroupContainerRe.FindString(string(b)); m != "" {
|
||||
return m
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func isProtectedUnit(id, name string) bool {
|
||||
return id == AgentUnit || name == strings.TrimSuffix(AgentUnit, ".service")
|
||||
}
|
||||
|
||||
func controlPlatform(ctx context.Context, kind, id, action string) error {
|
||||
var cmd *exec.Cmd
|
||||
switch kind {
|
||||
case "container":
|
||||
cmd = exec.CommandContext(ctx, "docker", action, id)
|
||||
case "unit":
|
||||
cmd = exec.CommandContext(ctx, "systemctl", action, id)
|
||||
default:
|
||||
return fmt.Errorf("unknown workload kind %q", kind)
|
||||
}
|
||||
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
if ctx.Err() == context.DeadlineExceeded {
|
||||
return fmt.Errorf("%s %s timed out after %s", action, id, controlTimeout)
|
||||
}
|
||||
return fmt.Errorf("%s %s: %s", action, id, strings.TrimSpace(string(out)))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
package workloads
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"strings"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/agent/internal/winexec"
|
||||
)
|
||||
|
||||
// AgentUnit is the service this agent runs as — the NSSM service name written
|
||||
// by installer/setup.ps1. Change one, change the other.
|
||||
const AgentUnit = "VantageAgent"
|
||||
|
||||
// A Windows agent is never itself in a container; the Linux build reads
|
||||
// /proc/self/cgroup, and there is no equivalent question to ask here.
|
||||
var ownContainerID = ""
|
||||
|
||||
// Windows service names are case-insensitive, so the comparison must be too.
|
||||
func isProtectedUnit(id, name string) bool {
|
||||
return strings.EqualFold(id, AgentUnit) || strings.EqualFold(name, AgentUnit)
|
||||
}
|
||||
|
||||
func controlPlatform(ctx context.Context, kind, id, action string) error {
|
||||
switch kind {
|
||||
case "container":
|
||||
// Docker behaves identically on Windows, so this path is shared in
|
||||
// spirit with the Linux one rather than routed through PowerShell.
|
||||
cmd := exec.CommandContext(ctx, "docker", action, id)
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
if ctx.Err() == context.DeadlineExceeded {
|
||||
return fmt.Errorf("%s %s timed out after %s", action, id, controlTimeout)
|
||||
}
|
||||
return fmt.Errorf("%s %s: %s", action, id, strings.TrimSpace(string(out)))
|
||||
}
|
||||
return nil
|
||||
|
||||
case "unit":
|
||||
// -Force is required: Stop-Service without it refuses outright when
|
||||
// another service depends on the target, and that refusal reads to an
|
||||
// operator as a silent no-op.
|
||||
//
|
||||
// sc.exe is avoided because it returns before the operation completes,
|
||||
// which turns a timeout into a false success.
|
||||
var verb string
|
||||
switch action {
|
||||
case "start":
|
||||
verb = "Start-Service"
|
||||
case "stop":
|
||||
verb = "Stop-Service"
|
||||
case "restart":
|
||||
verb = "Restart-Service"
|
||||
default:
|
||||
return fmt.Errorf("unknown action %q", action)
|
||||
}
|
||||
|
||||
script := "$ErrorActionPreference='Stop'\n" + verb + " -Name " + psQuote(id)
|
||||
if action != "start" {
|
||||
script += " -Force"
|
||||
}
|
||||
|
||||
if _, err := winexec.Run(ctx, script); err != nil {
|
||||
if ctx.Err() == context.DeadlineExceeded {
|
||||
return fmt.Errorf("%s %s timed out after %s", action, id, controlTimeout)
|
||||
}
|
||||
return fmt.Errorf("%s %s: %w", action, id, err)
|
||||
}
|
||||
return nil
|
||||
|
||||
default:
|
||||
return fmt.Errorf("unknown workload kind %q", kind)
|
||||
}
|
||||
}
|
||||
@@ -2,9 +2,6 @@ package workloads
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
@@ -35,26 +32,12 @@ func Logs(ctx context.Context, kind, id string, tail int) (string, bool, error)
|
||||
ctx, cancel := context.WithTimeout(ctx, logTimeout)
|
||||
defer cancel()
|
||||
|
||||
var cmd *exec.Cmd
|
||||
switch kind {
|
||||
case "container":
|
||||
cmd = exec.CommandContext(ctx, "docker", "logs",
|
||||
"--tail", strconv.Itoa(tail), "--timestamps", id)
|
||||
case "unit":
|
||||
cmd = exec.CommandContext(ctx, "journalctl", "-u", id,
|
||||
"-n", strconv.Itoa(tail), "--no-pager", "--output=short-iso")
|
||||
default:
|
||||
return "", false, fmt.Errorf("unknown workload kind %q", kind)
|
||||
out, err := logsPlatform(ctx, kind, id, tail)
|
||||
if err != nil {
|
||||
return "", false, err
|
||||
}
|
||||
|
||||
// docker logs writes container stderr to our stderr, so both streams must
|
||||
// be captured or half the output silently disappears.
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err != nil && len(out) == 0 {
|
||||
return "", false, fmt.Errorf("read logs for %s: %s", id, errText(err))
|
||||
}
|
||||
|
||||
text, truncated := capLog(string(out))
|
||||
text, truncated := capLog(out)
|
||||
return text, truncated, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
package workloads
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"strconv"
|
||||
)
|
||||
|
||||
func logsPlatform(ctx context.Context, kind, id string, tail int) (string, error) {
|
||||
var cmd *exec.Cmd
|
||||
switch kind {
|
||||
case "container":
|
||||
cmd = exec.CommandContext(ctx, "docker", "logs",
|
||||
"--tail", strconv.Itoa(tail), "--timestamps", id)
|
||||
case "unit":
|
||||
cmd = exec.CommandContext(ctx, "journalctl", "-u", id,
|
||||
"-n", strconv.Itoa(tail), "--no-pager", "--output=short-iso")
|
||||
default:
|
||||
return "", fmt.Errorf("unknown workload kind %q", kind)
|
||||
}
|
||||
|
||||
// docker logs writes container stderr to our stderr, so both streams must
|
||||
// be captured or half the output silently disappears.
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err != nil && len(out) == 0 {
|
||||
return "", fmt.Errorf("read logs for %s: %s", id, errText(err))
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
package workloads
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"strconv"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/agent/internal/winexec"
|
||||
)
|
||||
|
||||
func logsPlatform(ctx context.Context, kind, id string, tail int) (string, error) {
|
||||
switch kind {
|
||||
case "container":
|
||||
cmd := exec.CommandContext(ctx, "docker", "logs",
|
||||
"--tail", strconv.Itoa(tail), "--timestamps", id)
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err != nil && len(out) == 0 {
|
||||
return "", fmt.Errorf("read logs for %s: %s", id, errText(err))
|
||||
}
|
||||
return string(out), nil
|
||||
|
||||
case "unit":
|
||||
display := serviceDisplayName(ctx, id)
|
||||
|
||||
// Timestamps are formatted PowerShell-side rather than left to
|
||||
// ConvertTo-Json, whose DateTime rendering differs between PowerShell
|
||||
// versions — one of them emits /Date(1699...)/.
|
||||
//
|
||||
// $ErrorActionPreference = 'SilentlyContinue' because Get-WinEvent
|
||||
// treats "no events matched" as a terminating error, and a quiet
|
||||
// service is normal.
|
||||
names := psQuote(id)
|
||||
if display != "" && display != id {
|
||||
names += "," + psQuote(display)
|
||||
}
|
||||
names += "," + psQuote(scmProvider)
|
||||
|
||||
// ProviderName includes the host-wide Service Control Manager, so a
|
||||
// -MaxEvents cap of exactly tail would apply to the combined stream
|
||||
// before parseEvents narrows SCM rows down to this service — on a
|
||||
// host with busy service churn the target's own events could be
|
||||
// squeezed out of the window entirely. Over-fetch instead, hard-capped
|
||||
// so a pathological host cannot pull an unbounded batch across the
|
||||
// wire, and let parseEvents trim to the last tail lines after
|
||||
// filtering.
|
||||
fetch := tail * 5
|
||||
if fetch > 2500 {
|
||||
fetch = 2500
|
||||
}
|
||||
|
||||
script := `
|
||||
$ErrorActionPreference = 'SilentlyContinue'
|
||||
$rows = Get-WinEvent -FilterHashtable @{LogName='System','Application'; ProviderName=@(` + names + `)} ` +
|
||||
`-MaxEvents ` + strconv.Itoa(fetch) + ` |
|
||||
ForEach-Object {
|
||||
[pscustomobject]@{
|
||||
t = $_.TimeCreated.ToUniversalTime().ToString('o')
|
||||
l = [string]$_.LevelDisplayName
|
||||
p = [string]$_.ProviderName
|
||||
m = [string]$_.Message
|
||||
}
|
||||
}
|
||||
ConvertTo-Json -InputObject @($rows) -Depth 3 -Compress
|
||||
`
|
||||
|
||||
out, err := winexec.Run(ctx, script)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("read events for %s: %w", id, err)
|
||||
}
|
||||
return parseEvents(out, id, display, tail)
|
||||
|
||||
default:
|
||||
return "", fmt.Errorf("unknown workload kind %q", kind)
|
||||
}
|
||||
}
|
||||
|
||||
// serviceDisplayName resolves a service's display name, which is what Service
|
||||
// Control Manager events name it by. An empty answer is fine — the filter then
|
||||
// matches on the service name alone.
|
||||
func serviceDisplayName(ctx context.Context, id string) string {
|
||||
out, err := winexec.Run(ctx,
|
||||
"$ErrorActionPreference='SilentlyContinue'\n"+
|
||||
"(Get-Service -Name "+psQuote(id)+").DisplayName")
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
return trimLine(out)
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
package workloads
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/agent/internal/winexec"
|
||||
)
|
||||
|
||||
const servicesTimeout = 60 * time.Second
|
||||
|
||||
const servicesScript = `
|
||||
$ErrorActionPreference = 'Stop'
|
||||
$svcs = Get-CimInstance Win32_Service | ForEach-Object {
|
||||
[pscustomobject]@{
|
||||
Name = $_.Name
|
||||
DisplayName = $_.DisplayName
|
||||
State = $_.State
|
||||
StartMode = $_.StartMode
|
||||
PathName = $_.PathName
|
||||
ExitCode = $_.ExitCode
|
||||
}
|
||||
}
|
||||
ConvertTo-Json -InputObject @($svcs) -Depth 3 -Compress
|
||||
`
|
||||
|
||||
// collectUnits enumerates Windows services. The bool and string it returns are
|
||||
// the same SystemdOK / SystemdError pair the Linux collector fills: the wire
|
||||
// shape is shared, and the UI words it per platform.
|
||||
func collectUnits(ctx context.Context) ([]Workload, bool, string) {
|
||||
ctx, cancel := context.WithTimeout(ctx, servicesTimeout)
|
||||
defer cancel()
|
||||
|
||||
out, err := winexec.Run(ctx, servicesScript)
|
||||
if err != nil {
|
||||
return nil, false, "Win32_Service query failed: " + err.Error()
|
||||
}
|
||||
|
||||
systemRoot := os.Getenv("SystemRoot")
|
||||
if systemRoot == "" {
|
||||
systemRoot = `C:\Windows`
|
||||
}
|
||||
|
||||
wls, err := parseServices(out, systemRoot)
|
||||
if err != nil {
|
||||
return nil, false, "Win32_Service output could not be read: " + err.Error()
|
||||
}
|
||||
return wls, true, ""
|
||||
}
|
||||
@@ -14,10 +14,10 @@ const systemdTimeout = 30 * time.Second
|
||||
// ten anyone cares about.
|
||||
var excludedPrefixes = []string{"systemd-", "user@", "user-", "session-", "init.scope"}
|
||||
|
||||
// collectSystemd enumerates services in two passes, because "running or
|
||||
// collectUnits enumerates services in two passes, because "running or
|
||||
// failed" and "enabled but stopped" are different questions — and an enabled
|
||||
// unit that is not running is exactly the one worth seeing.
|
||||
func collectSystemd(ctx context.Context) ([]Workload, bool, string) {
|
||||
func collectUnits(ctx context.Context) ([]Workload, bool, string) {
|
||||
if _, err := exec.LookPath("systemctl"); err != nil {
|
||||
return nil, false, ""
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
//go:build !linux && !windows
|
||||
|
||||
// The build constraint is load-bearing — see updates_other.go.
|
||||
package workloads
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
)
|
||||
|
||||
var ownContainerID = ""
|
||||
|
||||
func collectUnits(context.Context) ([]Workload, bool, string) { return nil, false, "" }
|
||||
func isProtectedUnit(string, string) bool { return false }
|
||||
|
||||
func controlPlatform(context.Context, string, string, string) error {
|
||||
return fmt.Errorf("workload control is not supported on this platform")
|
||||
}
|
||||
|
||||
func logsPlatform(context.Context, string, string, int) (string, error) {
|
||||
return "", fmt.Errorf("workload logs are not supported on this platform")
|
||||
}
|
||||
@@ -0,0 +1,224 @@
|
||||
package workloads
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// winService is one row of Get-CimInstance Win32_Service.
|
||||
//
|
||||
// Win32_Service rather than Get-Service: Get-Service exposes neither PathName
|
||||
// nor StartMode, and the filter below needs both.
|
||||
type winService struct {
|
||||
Name string `json:"Name"`
|
||||
DisplayName string `json:"DisplayName"`
|
||||
State string `json:"State"`
|
||||
StartMode string `json:"StartMode"`
|
||||
PathName string `json:"PathName"`
|
||||
ExitCode int `json:"ExitCode"`
|
||||
}
|
||||
|
||||
// exitCodeNeverStarted is ERROR_SERVICE_NEVER_STARTED. A stopped service
|
||||
// carrying it has not failed — it has not run since boot — and painting that
|
||||
// red would cry wolf on every host.
|
||||
const exitCodeNeverStarted = 1077
|
||||
|
||||
// servicePath extracts the executable from a Win32_Service PathName.
|
||||
//
|
||||
// A naive split on whitespace misfiles a substantial share of a real fleet:
|
||||
// `"C:\Program Files\X\x.exe" -service` is one path and one argument.
|
||||
func servicePath(pathName string) string {
|
||||
s := strings.TrimSpace(pathName)
|
||||
if s == "" {
|
||||
return ""
|
||||
}
|
||||
if s[0] == '"' {
|
||||
if end := strings.IndexByte(s[1:], '"'); end >= 0 {
|
||||
return s[1 : 1+end]
|
||||
}
|
||||
// No closing quote: a malformed or truncated PathName. Fall back to
|
||||
// the unquoted handling below on the text after the opening quote,
|
||||
// so this yields a bare path rather than a path plus trailing
|
||||
// argument text.
|
||||
s = s[1:]
|
||||
}
|
||||
if i := exeBoundaryIndex(s); i >= 0 {
|
||||
return s[:i+len(".exe")]
|
||||
}
|
||||
if i := strings.IndexAny(s, " \t"); i >= 0 {
|
||||
return s[:i]
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// exeBoundaryIndex finds the first ".exe" (case-insensitive) in s that
|
||||
// actually ends the executable name — followed by end-of-string, whitespace,
|
||||
// or a double quote — rather than continuing into a longer segment such as
|
||||
// ".exec". It returns -1 when no such occurrence exists, so a path like
|
||||
// `C:\Program Files\Ad.exec\tool.com -flag` is not misparsed by matching the
|
||||
// ".exe" inside "Ad.exec" and silently dropping the real filename.
|
||||
func exeBoundaryIndex(s string) int {
|
||||
lower := strings.ToLower(s)
|
||||
from := 0
|
||||
for {
|
||||
rel := strings.Index(lower[from:], ".exe")
|
||||
if rel < 0 {
|
||||
return -1
|
||||
}
|
||||
idx := from + rel
|
||||
end := idx + len(".exe")
|
||||
if end == len(s) || s[end] == ' ' || s[end] == '\t' || s[end] == '"' {
|
||||
return idx
|
||||
}
|
||||
from = idx + 1
|
||||
}
|
||||
}
|
||||
|
||||
// parseServices turns the collector's JSON into workloads.
|
||||
//
|
||||
// systemRoot is a parameter rather than an environment read so this is testable
|
||||
// off Windows. The caller passes %SystemRoot%.
|
||||
//
|
||||
// The filter mirrors the systemd collector's intent: show what an operator
|
||||
// installed, and show what is meant to be up but is not. Services under
|
||||
// %SystemRoot%\System32 are the platform's own, and a typical host has well
|
||||
// over a hundred of them.
|
||||
func parseServices(jsonText, systemRoot string) ([]Workload, error) {
|
||||
s := strings.TrimSpace(jsonText)
|
||||
if s == "" || s == "null" {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
var rows []winService
|
||||
if err := json.Unmarshal([]byte(s), &rows); err != nil {
|
||||
var one winService
|
||||
if err2 := json.Unmarshal([]byte(s), &one); err2 != nil {
|
||||
return nil, err
|
||||
}
|
||||
rows = []winService{one}
|
||||
}
|
||||
|
||||
sys32 := strings.ToLower(strings.TrimRight(systemRoot, `\`) + `\system32\`)
|
||||
|
||||
var wls []Workload
|
||||
for _, r := range rows {
|
||||
if p := strings.ToLower(servicePath(r.PathName)); p != "" && strings.HasPrefix(p, sys32) {
|
||||
continue
|
||||
}
|
||||
|
||||
running := strings.EqualFold(r.State, "Running")
|
||||
failed := !running && r.ExitCode != 0 && r.ExitCode != exitCodeNeverStarted
|
||||
auto := strings.HasPrefix(strings.ToLower(r.StartMode), "auto")
|
||||
if !running && !failed && !auto {
|
||||
continue
|
||||
}
|
||||
|
||||
// The wire shape is shared with the systemd collector — both report
|
||||
// under kind "unit" — so the state word has to be too, or the UI
|
||||
// (which colours and filters on it, and does so before it knows
|
||||
// which platform sent the row) needs two vocabularies for one kind.
|
||||
// running/stopped/failed become active/inactive/failed to match.
|
||||
state := "inactive"
|
||||
switch {
|
||||
case running:
|
||||
state = "active"
|
||||
case failed:
|
||||
state = "failed"
|
||||
}
|
||||
|
||||
name := r.DisplayName
|
||||
if name == "" {
|
||||
name = r.Name
|
||||
}
|
||||
|
||||
wls = append(wls, Workload{
|
||||
Kind: "unit",
|
||||
ID: r.Name,
|
||||
Name: name,
|
||||
State: state,
|
||||
})
|
||||
}
|
||||
return wls, nil
|
||||
}
|
||||
|
||||
// psQuote renders a Go string as a PowerShell single-quoted literal. Single
|
||||
// quotes suppress every form of expansion, so the only character needing an
|
||||
// escape is the quote itself, which is doubled.
|
||||
func psQuote(s string) string {
|
||||
return "'" + strings.ReplaceAll(s, "'", "''") + "'"
|
||||
}
|
||||
|
||||
// scmProvider is the provider every service's start and stop is logged under,
|
||||
// host-wide.
|
||||
const scmProvider = "Service Control Manager"
|
||||
|
||||
type winEvent struct {
|
||||
T string `json:"t"`
|
||||
L string `json:"l"`
|
||||
P string `json:"p"`
|
||||
M string `json:"m"`
|
||||
}
|
||||
|
||||
// parseEvents renders Get-WinEvent output as text in the shape journalctl
|
||||
// --output=short-iso produces, so the log dialog needs no per-platform
|
||||
// rendering: "<timestamp> <level> <message>", oldest first.
|
||||
//
|
||||
// The caller over-fetches from Get-WinEvent because the ProviderName filter
|
||||
// includes the host-wide Service Control Manager, and a -MaxEvents cap
|
||||
// applied before SCM rows are narrowed down to this service would squeeze the
|
||||
// target's own events out of the window on a host with busy service churn.
|
||||
// tail is therefore applied here, AFTER filtering and AFTER the oldest-first
|
||||
// reversal, keeping the last tail lines — the most recent lines are the ones
|
||||
// worth keeping, matching capLog's front-trim reasoning in the shared
|
||||
// logs.go.
|
||||
func parseEvents(jsonText, serviceName, displayName string, tail int) (string, error) {
|
||||
s := strings.TrimSpace(jsonText)
|
||||
if s == "" || s == "null" {
|
||||
return "", nil
|
||||
}
|
||||
|
||||
var rows []winEvent
|
||||
if err := json.Unmarshal([]byte(s), &rows); err != nil {
|
||||
var one winEvent
|
||||
if err2 := json.Unmarshal([]byte(s), &one); err2 != nil {
|
||||
return "", err
|
||||
}
|
||||
rows = []winEvent{one}
|
||||
}
|
||||
|
||||
var lines []string
|
||||
for _, e := range rows {
|
||||
if strings.EqualFold(e.P, scmProvider) {
|
||||
if !strings.Contains(e.M, serviceName) &&
|
||||
(displayName == "" || !strings.Contains(e.M, displayName)) {
|
||||
continue
|
||||
}
|
||||
}
|
||||
// Collapse every newline form, not just "\r\n": a message containing a
|
||||
// bare "\n" would otherwise still break the one-line-per-event shape
|
||||
// this renders for the log dialog, and undercount the tail trim above.
|
||||
msg := strings.TrimSpace(strings.NewReplacer("\r\n", " ", "\r", " ", "\n", " ").Replace(e.M))
|
||||
lines = append(lines, e.T+" "+e.L+" "+msg)
|
||||
}
|
||||
|
||||
// Get-WinEvent is newest-first. Reverse it.
|
||||
for i, j := 0, len(lines)-1; i < j; i, j = i+1, j-1 {
|
||||
lines[i], lines[j] = lines[j], lines[i]
|
||||
}
|
||||
|
||||
if tail > 0 && len(lines) > tail {
|
||||
lines = lines[len(lines)-tail:]
|
||||
}
|
||||
|
||||
return strings.Join(lines, "\n"), nil
|
||||
}
|
||||
|
||||
// trimLine reduces single-value PowerShell output to its first non-empty line.
|
||||
func trimLine(s string) string {
|
||||
for _, l := range strings.Split(s, "\n") {
|
||||
if t := strings.TrimSpace(l); t != "" {
|
||||
return t
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
@@ -0,0 +1,207 @@
|
||||
package workloads
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestServicePath(t *testing.T) {
|
||||
cases := []struct{ in, want string }{
|
||||
{`"C:\Program Files\Contoso\svc.exe" -service`, `C:\Program Files\Contoso\svc.exe`},
|
||||
{`C:\WINDOWS\system32\svchost.exe -k netsvcs`, `C:\WINDOWS\system32\svchost.exe`},
|
||||
{`C:\Vantage\vantage-agent.exe`, `C:\Vantage\vantage-agent.exe`},
|
||||
{`"C:\no\args.exe"`, `C:\no\args.exe`},
|
||||
{``, ``},
|
||||
// ".exe" appearing inside an earlier segment ("Ad.exec") must not be
|
||||
// treated as the end of the executable — that would drop the real
|
||||
// filename and arguments.
|
||||
{`C:\Program Files\Ad.exec\tool.com -flag`, `C:\Program`},
|
||||
// An unterminated quote falls back to the unquoted handling on the
|
||||
// text after the opening quote, yielding a bare path rather than a
|
||||
// path plus trailing argument text.
|
||||
{`"C:\Program Files\Contoso\svc.exe -service`, `C:\Program Files\Contoso\svc.exe`},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := servicePath(c.in); got != c.want {
|
||||
t.Errorf("servicePath(%q) = %q, want %q", c.in, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseServicesFilters(t *testing.T) {
|
||||
in := `[
|
||||
{"Name":"Contoso","DisplayName":"Contoso Broker","State":"Running","StartMode":"Auto","PathName":"\"C:\\Program Files\\Contoso\\svc.exe\" -service","ExitCode":0},
|
||||
{"Name":"Themes","DisplayName":"Themes","State":"Running","StartMode":"Auto","PathName":"C:\\WINDOWS\\system32\\svchost.exe -k netsvcs","ExitCode":0},
|
||||
{"Name":"Fabrikam","DisplayName":"Fabrikam Sync","State":"Stopped","StartMode":"Auto","PathName":"C:\\Fabrikam\\sync.exe","ExitCode":0},
|
||||
{"Name":"Northwind","DisplayName":"Northwind Poller","State":"Stopped","StartMode":"Manual","PathName":"C:\\Northwind\\poll.exe","ExitCode":0},
|
||||
{"Name":"Crashed","DisplayName":"Crashed Thing","State":"Stopped","StartMode":"Auto","PathName":"C:\\Crashed\\c.exe","ExitCode":1067}
|
||||
]`
|
||||
|
||||
got, err := parseServices(in, `C:\WINDOWS`)
|
||||
if err != nil {
|
||||
t.Fatalf("parseServices: %v", err)
|
||||
}
|
||||
|
||||
byID := map[string]Workload{}
|
||||
for _, w := range got {
|
||||
byID[w.ID] = w
|
||||
}
|
||||
|
||||
// The OS's own svchost service is dropped; a manual, stopped, never-failed
|
||||
// service is nobody's business either.
|
||||
if _, ok := byID["Themes"]; ok {
|
||||
t.Error("Themes (under %SystemRoot%) should be filtered out")
|
||||
}
|
||||
if _, ok := byID["Northwind"]; ok {
|
||||
t.Error("stopped Manual service should be filtered out")
|
||||
}
|
||||
if len(got) != 3 {
|
||||
t.Fatalf("got %d workloads, want 3: %+v", len(got), got)
|
||||
}
|
||||
|
||||
if w := byID["Contoso"]; w.Kind != "unit" || w.Name != "Contoso Broker" || w.State != "active" {
|
||||
t.Errorf("Contoso = %+v", w)
|
||||
}
|
||||
// Enabled but not running is exactly the row worth seeing.
|
||||
if byID["Fabrikam"].State != "inactive" {
|
||||
t.Errorf("Fabrikam state = %q, want inactive", byID["Fabrikam"].State)
|
||||
}
|
||||
// A non-zero exit code on a stopped service is a crash, not a clean stop.
|
||||
if byID["Crashed"].State != "failed" {
|
||||
t.Errorf("Crashed state = %q, want failed", byID["Crashed"].State)
|
||||
}
|
||||
}
|
||||
|
||||
// 1077 means "no attempt to start since boot" — a clean stopped service, not a
|
||||
// failure, and reporting it red would cry wolf on every host.
|
||||
func TestParseServicesExitCode1077(t *testing.T) {
|
||||
in := `[{"Name":"Idle","DisplayName":"Idle","State":"Stopped","StartMode":"Auto","PathName":"C:\\Idle\\i.exe","ExitCode":1077}]`
|
||||
got, err := parseServices(in, `C:\WINDOWS`)
|
||||
if err != nil {
|
||||
t.Fatalf("parseServices: %v", err)
|
||||
}
|
||||
if len(got) != 1 || got[0].State != "inactive" {
|
||||
t.Fatalf("got %+v, want one inactive workload", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseServicesSingleObjectAndEmpty(t *testing.T) {
|
||||
one := `{"Name":"Solo","DisplayName":"Solo","State":"Running","StartMode":"Auto","PathName":"C:\\Solo\\s.exe","ExitCode":0}`
|
||||
got, err := parseServices(one, `C:\WINDOWS`)
|
||||
if err != nil || len(got) != 1 || got[0].State != "active" {
|
||||
t.Fatalf("single object: got %+v, err %v", got, err)
|
||||
}
|
||||
|
||||
for _, in := range []string{"", "[]", "null"} {
|
||||
got, err := parseServices(in, `C:\WINDOWS`)
|
||||
if err != nil || len(got) != 0 {
|
||||
t.Fatalf("parseServices(%q) = %+v, err %v", in, got, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPSQuote(t *testing.T) {
|
||||
if got := psQuote(`it's`); got != `'it''s'` {
|
||||
t.Fatalf("psQuote = %s", got)
|
||||
}
|
||||
if got := psQuote(`plain`); got != `'plain'` {
|
||||
t.Fatalf("psQuote = %s", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseEventsFormatsAndOrders(t *testing.T) {
|
||||
// Get-WinEvent returns newest first; journalctl --output=short-iso returns
|
||||
// oldest first, and the log dialog and capLog's front-trim both assume the
|
||||
// most recent line is at the bottom.
|
||||
in := `[
|
||||
{"t":"2026-08-13T10:22:31.0000000Z","l":"Error","p":"Contoso","m":"broker died"},
|
||||
{"t":"2026-08-13T10:22:03.0000000Z","l":"Information","p":"Contoso","m":"broker starting"}
|
||||
]`
|
||||
|
||||
got, err := parseEvents(in, "Contoso", "Contoso Broker", 500)
|
||||
if err != nil {
|
||||
t.Fatalf("parseEvents: %v", err)
|
||||
}
|
||||
|
||||
want := "2026-08-13T10:22:03.0000000Z Information broker starting\n" +
|
||||
"2026-08-13T10:22:31.0000000Z Error broker died"
|
||||
if got != want {
|
||||
t.Fatalf("parseEvents =\n%q\nwant\n%q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// A message containing a bare "\n" (no carriage return) must still collapse to
|
||||
// one line, or it silently multiplies into several output lines and throws
|
||||
// off the tail trim's count.
|
||||
func TestParseEventsCollapsesBareLF(t *testing.T) {
|
||||
in := `[{"t":"2026-08-13T10:00:00Z","l":"Error","p":"Contoso","m":"broker died\nstack trace here"}]`
|
||||
|
||||
got, err := parseEvents(in, "Contoso", "Contoso Broker", 500)
|
||||
if err != nil {
|
||||
t.Fatalf("parseEvents: %v", err)
|
||||
}
|
||||
if strings.Count(got, "\n") != 0 {
|
||||
t.Fatalf("parseEvents did not collapse bare LF into one line: %q", got)
|
||||
}
|
||||
want := "2026-08-13T10:00:00Z Error broker died stack trace here"
|
||||
if got != want {
|
||||
t.Fatalf("parseEvents =\n%q\nwant\n%q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// Service Control Manager logs every service on the host under one provider, so
|
||||
// its rows must be filtered down to the target or the log is somebody else's.
|
||||
func TestParseEventsFiltersOtherServicesSCM(t *testing.T) {
|
||||
in := `[
|
||||
{"t":"2026-08-13T10:00:00Z","l":"Information","p":"Service Control Manager","m":"The Print Spooler service entered the running state."},
|
||||
{"t":"2026-08-13T10:00:01Z","l":"Information","p":"Service Control Manager","m":"The Contoso Broker service entered the running state."}
|
||||
]`
|
||||
|
||||
got, err := parseEvents(in, "Contoso", "Contoso Broker", 500)
|
||||
if err != nil {
|
||||
t.Fatalf("parseEvents: %v", err)
|
||||
}
|
||||
if strings.Contains(got, "Print Spooler") {
|
||||
t.Errorf("another service's SCM event leaked in:\n%s", got)
|
||||
}
|
||||
if !strings.Contains(got, "Contoso Broker") {
|
||||
t.Errorf("the target's SCM event was dropped:\n%s", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A service that has logged nothing is normal. An error there would read as a
|
||||
// broken feature.
|
||||
func TestParseEventsEmpty(t *testing.T) {
|
||||
for _, in := range []string{"", "[]", "null"} {
|
||||
got, err := parseEvents(in, "Contoso", "Contoso Broker", 500)
|
||||
if err != nil || got != "" {
|
||||
t.Fatalf("parseEvents(%q) = %q, err %v", in, got, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The over-fetch in logs_windows.go can return more events than the caller
|
||||
// asked for once SCM rows are filtered down to the target; parseEvents must
|
||||
// keep the most RECENT tail lines, not the oldest, matching capLog's
|
||||
// front-trim reasoning in the shared logs.go.
|
||||
func TestParseEventsTrimsToTailKeepingMostRecent(t *testing.T) {
|
||||
in := `[
|
||||
{"t":"2026-08-13T10:00:06Z","l":"Information","p":"Contoso","m":"event 6"},
|
||||
{"t":"2026-08-13T10:00:05Z","l":"Information","p":"Contoso","m":"event 5"},
|
||||
{"t":"2026-08-13T10:00:04Z","l":"Information","p":"Contoso","m":"event 4"},
|
||||
{"t":"2026-08-13T10:00:03Z","l":"Information","p":"Contoso","m":"event 3"},
|
||||
{"t":"2026-08-13T10:00:02Z","l":"Information","p":"Contoso","m":"event 2"},
|
||||
{"t":"2026-08-13T10:00:01Z","l":"Information","p":"Contoso","m":"event 1"}
|
||||
]`
|
||||
|
||||
got, err := parseEvents(in, "Contoso", "Contoso Broker", 2)
|
||||
if err != nil {
|
||||
t.Fatalf("parseEvents: %v", err)
|
||||
}
|
||||
|
||||
want := "2026-08-13T10:00:05Z Information event 5\n" +
|
||||
"2026-08-13T10:00:06Z Information event 6"
|
||||
if got != want {
|
||||
t.Fatalf("parseEvents =\n%q\nwant\n%q", got, want)
|
||||
}
|
||||
}
|
||||
@@ -4,7 +4,6 @@ import (
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"runtime"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -19,15 +18,12 @@ type Result struct {
|
||||
SystemdError string
|
||||
}
|
||||
|
||||
// Collect enumerates every workload on this host. Linux only.
|
||||
// Collect enumerates every workload on this host: containers from Docker, and
|
||||
// units from systemd on Linux or the service control manager on Windows.
|
||||
func Collect(ctx context.Context) Result {
|
||||
if runtime.GOOS != "linux" {
|
||||
return Result{}
|
||||
}
|
||||
|
||||
var r Result
|
||||
containers, dockerOK, dockerErr := collectDocker(ctx)
|
||||
units, systemdOK, systemdErr := collectSystemd(ctx)
|
||||
units, systemdOK, systemdErr := collectUnits(ctx)
|
||||
|
||||
r.DockerOK, r.DockerError = dockerOK, dockerErr
|
||||
r.SystemdOK, r.SystemdError = systemdOK, systemdErr
|
||||
|
||||
@@ -1,30 +0,0 @@
|
||||
# Current cloud instance process
|
||||
|
||||
The current processs for creating cloud instances is incorrect.
|
||||
|
||||
At the moment the cloud instance process is the following:
|
||||
|
||||
- Customer goes to `https://vantage.hostxtra.co.uk/start` then fills in the form.
|
||||
- Customer is then sent and email to verify
|
||||
- Customer clicks the link and the instance is created in the DB.
|
||||
- Customer can then access the instance.
|
||||
|
||||
As the `/start` process is auto creating a new instance this should default to the free tier instance.
|
||||
|
||||
The issue is that this doesn't create an `account` and `admin_instance` on the admin side.
|
||||
|
||||
The Cloud instance creation / account creation needs to be restructured.
|
||||
|
||||
for context when I say `hq` I mean `admin`
|
||||
|
||||
- customer goes to `https://vantage.hostxtra.co.uk/start` and fills in the form.
|
||||
- This is where the HQ account is created.
|
||||
- The customer is then sent and email to verify their email address.
|
||||
- The customer can then access the HQ customer portal.
|
||||
- The customer can then create a free new instance in the HQ portal.
|
||||
- The cloud instance is created in the DB.
|
||||
- The HQ `account` and `admin_instance` is created and populated in the DB.
|
||||
- A `Free` License is created and attached to the instance.
|
||||
- The customer is then sent an email letting them know the instance has been created and when the license expires.
|
||||
- The customer will need to renew the license after expiry, if they are on a Free license.
|
||||
- This is so that unused instances can be cleaned up if no renew after a length of time has passed.
|
||||
@@ -2,5 +2,5 @@ apiVersion: v2
|
||||
name: vantage
|
||||
description: Helm chart for the Vantage stack (Redis, MongoDB, guacd, server, web)
|
||||
type: application
|
||||
version: 1.0.8
|
||||
version: 1.1.0
|
||||
appVersion: "1.0.8"
|
||||
|
||||
@@ -41,12 +41,9 @@ Ingress (Traefik):
|
||||
{{- range .Values.ingress.web.extraHosts }}
|
||||
https://{{ . }}
|
||||
{{- end }}
|
||||
{{- if .Values.ingress.api.enabled }}
|
||||
{{ join ", " .Values.ingress.api.paths }} go straight to the server; everything else to web.
|
||||
{{- else }}
|
||||
Everything goes to web, which proxies /api and /auth onward. Set
|
||||
ingress.api.enabled=true to route them at the edge instead.
|
||||
{{- end }}
|
||||
{{ join ", " .Values.ingress.api.paths }} go to the server; everything else to web.
|
||||
web proxies nothing, so those paths must be routed here or by a terminator
|
||||
in front of this ingress.
|
||||
{{- if .Values.ingress.grpc.enabled }}
|
||||
- Agents: {{ .Values.ingress.grpc.host }} (gRPC, h2c behind TLS)
|
||||
Agents dial server.env.grpcHost, currently {{ tpl .Values.server.env.grpcHost . }}.
|
||||
@@ -70,3 +67,13 @@ or add an Ingress on top of the -web and -server services.
|
||||
Quick access via port-forward, e.g.:
|
||||
kubectl port-forward svc/{{ .Release.Name }}-web {{ .Values.web.service.port }}:{{ .Values.web.service.port }}
|
||||
kubectl port-forward svc/{{ .Release.Name }}-server {{ .Values.server.service.httpPort }}:{{ .Values.server.service.httpPort }}
|
||||
{{- if not .Values.backup.enabled }}
|
||||
|
||||
No backups are scheduled. Vantage encrypts SSH private keys, vault secrets and
|
||||
SSO client secrets with KEY_ENCRYPTION_KEY, and that key is not stored anywhere
|
||||
but your own configuration — a database restored without it is permanently
|
||||
unreadable.
|
||||
|
||||
Set backup.enabled, backup.image and backup.pvcName, and store
|
||||
KEY_ENCRYPTION_KEY somewhere that survives this cluster.
|
||||
{{- end }}
|
||||
|
||||
@@ -72,6 +72,8 @@ both read it.
|
||||
value: {{ .Values.server.env.proxyAdvertiseHost | quote }}
|
||||
- name: PROXY_LISTEN_HOST
|
||||
value: {{ .Values.server.env.proxyListenHost | quote }}
|
||||
- name: TRUSTED_PROXIES
|
||||
value: {{ .Values.server.env.trustedProxies | quote }}
|
||||
{{- if eq .Values.server.env.deploymentType "cloud" }}
|
||||
- name: VANTAGE_DEPLOYMENT
|
||||
value: "cloud"
|
||||
@@ -83,3 +85,18 @@ both read it.
|
||||
fieldRef:
|
||||
fieldPath: status.podIP
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
vantage.backup.env renders the environment vantagectl needs.
|
||||
|
||||
It reads the SAME values the server does rather than taking its own, because a
|
||||
backup that connected to a different database, or stamped a fingerprint of a
|
||||
different key, than the deployment it is backing up would be worse than no
|
||||
backup: it would look like one.
|
||||
*/}}
|
||||
{{- define "vantage.backup.env" -}}
|
||||
- name: MONGO_URI
|
||||
value: {{ tpl .Values.server.env.mongoUri . | quote }}
|
||||
- name: KEY_ENCRYPTION_KEY
|
||||
value: {{ .Values.server.env.keyEncryptionKey | quote }}
|
||||
{{- end -}}
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
{{- if .Values.backup.enabled }}
|
||||
{{- if not .Values.backup.pvcName }}
|
||||
{{- fail "backup.enabled requires backup.pvcName: a backup needs somewhere durable to land, and the chart cannot guess where that is" }}
|
||||
{{- end }}
|
||||
{{- if not .Values.backup.image }}
|
||||
{{- fail "backup.enabled requires backup.image: the vantagectl image to run" }}
|
||||
{{- end }}
|
||||
apiVersion: batch/v1
|
||||
kind: CronJob
|
||||
metadata:
|
||||
name: {{ include "vantage.fullname" . }}-backup
|
||||
labels:
|
||||
{{- include "vantage.labels" . | nindent 4 }}
|
||||
app.kubernetes.io/component: backup
|
||||
spec:
|
||||
schedule: {{ .Values.backup.schedule | quote }}
|
||||
concurrencyPolicy: Forbid
|
||||
successfulJobsHistoryLimit: {{ .Values.backup.successfulJobsHistoryLimit }}
|
||||
failedJobsHistoryLimit: {{ .Values.backup.failedJobsHistoryLimit }}
|
||||
jobTemplate:
|
||||
spec:
|
||||
backoffLimit: 2
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
{{- include "vantage.labels" . | nindent 12 }}
|
||||
app.kubernetes.io/component: backup
|
||||
spec:
|
||||
restartPolicy: Never
|
||||
containers:
|
||||
- name: vantagectl
|
||||
image: {{ .Values.backup.image | quote }}
|
||||
args:
|
||||
- backup
|
||||
- --out
|
||||
- /backups
|
||||
{{- with .Values.backup.exclude }}
|
||||
- --exclude
|
||||
- {{ join "," . | quote }}
|
||||
{{- end }}
|
||||
env:
|
||||
# Referenced, never redeclared. A backup job holding its own
|
||||
# copy of KEY_ENCRYPTION_KEY is a second place for it to be
|
||||
# wrong, and the fingerprint it stamps would then be a
|
||||
# fingerprint of the wrong key.
|
||||
{{- include "vantage.backup.env" . | nindent 16 }}
|
||||
volumeMounts:
|
||||
- name: backups
|
||||
mountPath: /backups
|
||||
resources:
|
||||
{{- toYaml .Values.backup.resources | nindent 16 }}
|
||||
volumes:
|
||||
- name: backups
|
||||
persistentVolumeClaim:
|
||||
claimName: {{ .Values.backup.pvcName | quote }}
|
||||
{{- end }}
|
||||
@@ -2,16 +2,14 @@
|
||||
{{/*
|
||||
Two hostnames, because the two audiences arrive over different protocols.
|
||||
|
||||
Browsers reach the web host. What answers there depends on the path: with
|
||||
ingress.api.enabled, /api and /auth go straight to the server and everything
|
||||
else goes to `web`. Without it, everything goes to `web`, which proxies those
|
||||
prefixes onward itself (web/next.config.ts).
|
||||
Browsers reach the web host, and the path decides what answers: /api, /auth,
|
||||
/public, /install* and /update* go to the server, everything else to `web`.
|
||||
|
||||
Both work. Routing at the edge is one hop shorter and is what the Nginx Proxy
|
||||
Manager deployment in front of the Docker install already does, so leaving it
|
||||
off changes the shape of the request path between the two deployments. It is
|
||||
still off by default, because turning it on where `web` is the only thing with
|
||||
a public certificate would strand /api behind a route nobody can reach.
|
||||
That split is not optional and ingress.api.enabled defaults to true. `web`
|
||||
proxies nothing — it holds no address for the server at all — so with these
|
||||
paths absent the UI loads and every request it makes 404s against Next. The
|
||||
setting remains a value only so an installation terminating in front of this
|
||||
ingress can route the prefixes itself; it must be routed somewhere.
|
||||
|
||||
The web host is normally a wildcard — `*.vantage.example.com` — because that is
|
||||
the per-tenant instance namespace; APP_ROOT_LABEL resolves the instance from the
|
||||
@@ -35,6 +33,9 @@ its HTTP port too.
|
||||
{{- if and .Values.ingress.api.enabled (not $apiPaths) }}
|
||||
{{- fail "ingress.api.enabled requires at least one path in ingress.api.paths" }}
|
||||
{{- end }}
|
||||
{{- if not .Values.ingress.api.enabled }}
|
||||
{{- fail "ingress.api.enabled=false leaves /api, /auth and /public unrouted: web proxies nothing. Route those prefixes to the server at your own terminator, or leave this enabled." }}
|
||||
{{- end }}
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: Ingress
|
||||
metadata:
|
||||
|
||||
@@ -38,12 +38,10 @@ spec:
|
||||
image: "{{ .Values.web.image.repository }}:{{ .Values.web.image.tag }}"
|
||||
ports:
|
||||
- containerPort: {{ .Values.web.service.port }}
|
||||
env:
|
||||
- name: API_URL
|
||||
value: {{ tpl .Values.web.env.apiUrl . | quote }}
|
||||
# /healthz is served by this Next process; /api is rewritten to the
|
||||
# server, so a probe there would report the backend's health and keep
|
||||
# passing while this pod was wedged.
|
||||
# /healthz is served by this Next process. /api never reaches this
|
||||
# pod at all — the ingress routes it to the server — so there is no
|
||||
# backend address to configure and no probe here that could report
|
||||
# the backend's health by accident.
|
||||
startupProbe:
|
||||
httpGet:
|
||||
path: /healthz
|
||||
|
||||
@@ -63,6 +63,7 @@ server:
|
||||
appRootLabel: vantage
|
||||
proxyAdvertiseHost: "{{ .Release.Name }}-server"
|
||||
proxyListenHost: "0.0.0.0"
|
||||
trustedProxies: "10.0.0.0/8,172.16.0.0/12,192.168.0.0/16"
|
||||
persistence:
|
||||
enabled: false
|
||||
size: 1Gi
|
||||
@@ -78,8 +79,6 @@ web:
|
||||
service:
|
||||
type: ClusterIP
|
||||
port: 3000
|
||||
env:
|
||||
apiUrl: "http://{{ .Release.Name }}-server:8080"
|
||||
|
||||
ingress:
|
||||
enabled: false
|
||||
@@ -89,11 +88,14 @@ ingress:
|
||||
web:
|
||||
host: ""
|
||||
extraHosts: []
|
||||
# Not optional: web proxies nothing, so these prefixes reach the server
|
||||
# only through this ingress. Turning it off serves the UI with a dead API.
|
||||
api:
|
||||
enabled: false
|
||||
enabled: true
|
||||
paths:
|
||||
- /api/
|
||||
- /auth/
|
||||
- /public/
|
||||
- /update
|
||||
- /install
|
||||
- /update.ps1
|
||||
@@ -109,3 +111,25 @@ ingress:
|
||||
certResolver: ""
|
||||
|
||||
imagePullSecrets: []
|
||||
|
||||
# Scheduled backups.
|
||||
#
|
||||
# Off by default, deliberately. A backup with nowhere durable to land is a
|
||||
# false sense of safety, and the chart cannot know where that is — pvcName
|
||||
# must name a volume you have decided will outlive the cluster.
|
||||
#
|
||||
# There is no restore manifest here on purpose: a restore is an operator
|
||||
# decision with a confirmation attached, and must never be something a
|
||||
# `helm upgrade` can trigger. Run one as a `kubectl run` Job with
|
||||
# --confirm-db.
|
||||
backup:
|
||||
enabled: false
|
||||
schedule: "0 2 * * *"
|
||||
image: ""
|
||||
pvcName: ""
|
||||
# Collections to leave out. Recorded in each archive's manifest, so an
|
||||
# archive can never claim to be complete when it is not.
|
||||
exclude: []
|
||||
successfulJobsHistoryLimit: 3
|
||||
failedJobsHistoryLimit: 3
|
||||
resources: {}
|
||||
|
||||
@@ -47,6 +47,7 @@ services:
|
||||
KEY_ENCRYPTION_KEY: ${KEY_ENCRYPTION_KEY:-}
|
||||
GUACD_ADDR: guacd:4822
|
||||
PROXY_ADVERTISE_HOST: server
|
||||
TRUSTED_PROXIES: ${TRUSTED_PROXIES:-10.0.0.0/8,172.16.0.0/12,192.168.0.0/16}
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_healthy
|
||||
@@ -59,8 +60,10 @@ services:
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- 3000:3000
|
||||
environment:
|
||||
API_URL: ${API_URL:-http://server:8080}
|
||||
# No API_URL: web proxies nothing. The reverse proxy in front of this
|
||||
# deployment must route /api, /auth, /public, /install*, /update* to
|
||||
# server:8080 and everything else to web:3000. Reaching web:3000
|
||||
# directly serves the UI and every API call 404s.
|
||||
depends_on:
|
||||
- server
|
||||
volumes:
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,938 +0,0 @@
|
||||
# Instance Rename in Vantage HQ — Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Let an HQ customer (owner or admin) rename a cloud instance, which re-derives its slug and moves it to a new DNS host, with staff able to do the same without the cooldown.
|
||||
|
||||
**Architecture:** Slug derivation stays in `shared/provision`, beside the create path that already owns it. Admin reaches the control plane only through `cloudprov`, writing `instances` — a collection it already writes. Admin's own row (`admin_instances`) is updated second and carries the 24h cooldown timestamp, because the cooldown is admin's policy and the control plane has no opinion about it. The portal shows the new host and asks the customer to click through; it does not redirect.
|
||||
|
||||
**Note:** This repo has no automated test suite and the user has ruled out adding test files. Every task verifies by build, vet and (Task 8) manual exercise.
|
||||
|
||||
**Tech Stack:** Go 1.x (gin, mongo-driver v2), Next.js 16 App Router + TanStack Query + Tailwind 3 (`adminsite`).
|
||||
|
||||
**Spec:** `docs/superpowers/specs/2026-08-12-instance-rename-design.md`
|
||||
|
||||
## Global Constraints
|
||||
|
||||
- A licence binds an instance **UUID**, not a slug. A rename must not issue a licence, call Paddle, or touch `licenses`, `subscriptions` or `entitlements`.
|
||||
- Admin's control-plane write boundary is unchanged: `cloudprov` writes `instances` and `users` only. Do not add a write to any other control-plane collection.
|
||||
- Customer rename is **cloud only**. Self-hosted is refused with the existing `selfHostedRefusal` constant and HTTP **400**, matching `members.go`.
|
||||
- Cooldown for customers is **24 hours**, tracked by `admin_instances.renamed_at`. Staff bypass it and must **not** write `renamed_at`.
|
||||
- No `-2` suffix loop on rename. A taken slug is a refusal (`ErrSlugTaken` → HTTP 409).
|
||||
- No component in `adminsite` may carry a hex colour; use the existing token classes (`text-ink-2`, `text-ink-3`, `border-rule`, `text-accent`, `text-expired`, `bg-panel-2`).
|
||||
- The host domain used for display is `vantage.hostxtra.co.uk`, already hardcoded in `adminsite/components/InstanceRecord.tsx` and the customer instance page.
|
||||
- Commit messages follow the repo's existing style: `feat: Sentence case summary` / `fix: …` / `docs: …`.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Slug derivation and the control-plane rename
|
||||
|
||||
**Files:**
|
||||
- Modify: `shared/provision/instance.go`
|
||||
**Interfaces:**
|
||||
- Consumes: `BaseSlug(name string) (string, error)`, `ErrNameRejected` — both already in `shared/provision`.
|
||||
- Produces:
|
||||
- `provision.ErrSlugTaken` (`error`)
|
||||
- `provision.RenameSlug(name, currentSlug string) (string, error)`
|
||||
- `provision.RenameInstance(ctx context.Context, db *mongo.Database, instanceID, name string) (*models.Instance, error)`
|
||||
- `provision.RestoreInstanceIdentity(ctx context.Context, db *mongo.Database, instanceID, name, slug string) error`
|
||||
|
||||
Behaviour `RenameSlug` must have, verified by reading rather than by test (this
|
||||
repo has no Go test suite and the user has ruled out adding one):
|
||||
|
||||
| Input name | Current slug | Result |
|
||||
|---|---|---|
|
||||
| `Acme Ltd` | `acme` | `acme-ltd` |
|
||||
| `ACME!` | `acme` | `acme` — still derives to the current slug, so not a move |
|
||||
| `Acme` | `acme-2` | `acme` — a creation-time collision suffix derives from no name, so moving off it is a real move |
|
||||
| `ab` | any | `ErrNameRejected` |
|
||||
| `Admin` | any | `ErrNameRejected` (reserved) |
|
||||
| `!!!` | any | `ErrNameRejected` |
|
||||
| 50 `a`s | any | truncated to `MaxSlugLength`, exactly as `BaseSlug` truncates on create |
|
||||
|
||||
- [ ] **Step 1: Write the implementation**
|
||||
|
||||
Append to `shared/provision/instance.go`:
|
||||
|
||||
```go
|
||||
// ErrSlugTaken means the slug a new name derives to already belongs to another
|
||||
// instance.
|
||||
//
|
||||
// Rename refuses rather than appending a counter the way creation does. Creation
|
||||
// appends because the customer is waiting on an instance and any free slug will
|
||||
// do; a rename is a request for one specific host, and silently landing them on
|
||||
// "acme-2" answers a question they did not ask.
|
||||
var ErrSlugTaken = errors.New("slug taken")
|
||||
|
||||
// RenameSlug derives the slug a rename to name would move an instance to, given
|
||||
// the slug it holds now.
|
||||
//
|
||||
// It returns the current slug unchanged when the name still derives to it, so a
|
||||
// cosmetic edit — capitalisation, punctuation, a trailing "Ltd." — is not a move
|
||||
// and cannot collide with the instance's own slug.
|
||||
func RenameSlug(name, currentSlug string) (string, error) {
|
||||
base, err := BaseSlug(name)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("%w: %s", ErrNameRejected, err.Error())
|
||||
}
|
||||
if base == currentSlug {
|
||||
return currentSlug, nil
|
||||
}
|
||||
return base, nil
|
||||
}
|
||||
|
||||
// RenameInstance changes an instance's name and re-derives its slug from it.
|
||||
//
|
||||
// The count-then-update is racy on its own, and is safe for the same reason
|
||||
// CreateInstanceWithID's loop is: instances.slug carries a unique index, so a
|
||||
// lost race surfaces as a duplicate-key error. Unlike creation there is nothing
|
||||
// to retry with — the caller asked for one specific name — so it becomes
|
||||
// ErrSlugTaken. Do not remove the duplicate-key branch, and do not remove the
|
||||
// index.
|
||||
func RenameInstance(ctx context.Context, db *mongo.Database, instanceID, name string) (*models.Instance, error) {
|
||||
var inst models.Instance
|
||||
if err := db.Collection("instances").FindOne(ctx,
|
||||
bson.M{"instance_id": instanceID}).Decode(&inst); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
slug, err := RenameSlug(name, inst.Slug)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
if slug != inst.Slug {
|
||||
n, err := db.Collection("instances").CountDocuments(ctx, bson.M{
|
||||
"slug": slug,
|
||||
"instance_id": bson.M{"$ne": instanceID},
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if n > 0 {
|
||||
return nil, fmt.Errorf("%w: %s", ErrSlugTaken, slug)
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := db.Collection("instances").UpdateOne(ctx,
|
||||
bson.M{"instance_id": instanceID},
|
||||
bson.M{"$set": bson.M{"name": name, "slug": slug}}); err != nil {
|
||||
if mongo.IsDuplicateKeyError(err) {
|
||||
return nil, fmt.Errorf("%w: %s", ErrSlugTaken, slug)
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
|
||||
inst.Name = name
|
||||
inst.Slug = slug
|
||||
return &inst, nil
|
||||
}
|
||||
|
||||
// RestoreInstanceIdentity writes an exact name and slug back, unwinding a rename
|
||||
// whose caller-side bookkeeping then failed.
|
||||
//
|
||||
// It derives nothing. The values being restored may include a creation-time
|
||||
// collision suffix that no name derives to, so re-running RenameInstance with the
|
||||
// old name would not reproduce them.
|
||||
func RestoreInstanceIdentity(ctx context.Context, db *mongo.Database, instanceID, name, slug string) error {
|
||||
_, err := db.Collection("instances").UpdateOne(ctx,
|
||||
bson.M{"instance_id": instanceID},
|
||||
bson.M{"$set": bson.M{"name": name, "slug": slug}})
|
||||
return err
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Build and vet**
|
||||
|
||||
Run: `cd /go-projects/vantage && go build ./shared/... && go vet ./shared/provision/`
|
||||
Expected: clean.
|
||||
|
||||
- [ ] **Step 3: Commit**
|
||||
|
||||
```bash
|
||||
git add shared/provision/instance.go
|
||||
git commit -m "feat: Add instance rename to shared provisioning"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 2: Admin's row and the cloudprov wrappers
|
||||
|
||||
**Files:**
|
||||
- Modify: `admin/internal/models/models.go` (the `Instance` struct, ~line 129; constants block near `RenewWindow`, ~line 105)
|
||||
- Modify: `admin/internal/cloudprov/cloudprov.go`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `provision.RenameInstance`, `provision.RestoreInstanceIdentity` (Task 1).
|
||||
- Produces:
|
||||
- `models.RenameCooldown` (`time.Duration`)
|
||||
- `models.Instance.RenamedAt *time.Time` (bson `renamed_at`, json `renamed_at`)
|
||||
- `cloudprov.RenameInstance(ctx context.Context, instanceID, name string) (*sharedmodels.Instance, error)`
|
||||
- `cloudprov.RestoreInstanceIdentity(ctx context.Context, instanceID, name, slug string) error`
|
||||
|
||||
- [ ] **Step 1: Add the cooldown constant**
|
||||
|
||||
In `admin/internal/models/models.go`, directly beneath the `RenewWindow` block:
|
||||
|
||||
```go
|
||||
// RenameCooldown is how long a customer must wait between renames of one
|
||||
// instance.
|
||||
//
|
||||
// A rename moves the instance's DNS host and invalidates every saved link to it,
|
||||
// so this exists to make that a considered act rather than a slider. Staff are
|
||||
// not subject to it: a support conversation about a name is already a human
|
||||
// deciding.
|
||||
const RenameCooldown = 24 * time.Hour
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Add the field to `Instance`**
|
||||
|
||||
In the same file, inside the `Instance` struct, after `RelinkCount`:
|
||||
|
||||
```go
|
||||
// RenamedAt is when this instance last changed name, and backs the customer
|
||||
// rename cooldown. It is a pointer because absent means "never renamed"; a
|
||||
// zero time.Time would read as year 1 — an inert cooldown, but only by
|
||||
// accident. Staff renames deliberately leave it alone.
|
||||
RenamedAt *time.Time `bson:"renamed_at,omitempty" json:"renamed_at,omitempty"`
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Add the cloudprov wrappers**
|
||||
|
||||
Append to `admin/internal/cloudprov/cloudprov.go`:
|
||||
|
||||
```go
|
||||
// RenameInstance changes a cloud instance's name and moves it to the slug that
|
||||
// name derives to.
|
||||
//
|
||||
// It writes `instances` and nothing else, so admin's control-plane write
|
||||
// boundary is unchanged. It issues no licence: a licence binds the instance
|
||||
// UUID, which a rename never touches.
|
||||
func RenameInstance(ctx context.Context, instanceID, name string) (*sharedmodels.Instance, error) {
|
||||
return provision.RenameInstance(ctx, db.ControlDB(), instanceID, name)
|
||||
}
|
||||
|
||||
// RestoreInstanceIdentity puts an instance's previous name and slug back, for a
|
||||
// caller unwinding a rename whose admin-side write failed. Leaving the two
|
||||
// databases disagreeing would have HQ print a host that is not the host.
|
||||
func RestoreInstanceIdentity(ctx context.Context, instanceID, name, slug string) error {
|
||||
return provision.RestoreInstanceIdentity(ctx, db.ControlDB(), instanceID, name, slug)
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Build**
|
||||
|
||||
Run: `cd /go-projects/vantage && go build ./admin/... ./shared/...`
|
||||
Expected: clean build, no output.
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add admin/internal/models/models.go admin/internal/cloudprov/cloudprov.go
|
||||
git commit -m "feat: Add rename cooldown field and cloudprov rename"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 3: Customer rename endpoint
|
||||
|
||||
**Files:**
|
||||
- Modify: `admin/internal/api/customer.go` (add handler; `loginURLFor` at ~line 442 is already in this file)
|
||||
- Modify: `admin/internal/api/routes.go` (~line 77, beside the other `/instances/:id/*` customer routes)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `ownedInstance(c, id) (*models.Instance, bool)`, `selfHostedRefusal` (`members.go`), `loginURLFor(slug) string`, `cloudprov.RenameInstance`, `cloudprov.RestoreInstanceIdentity`, `models.RenameCooldown`, `provision.ErrSlugTaken`, `provision.ErrNameRejected`.
|
||||
- Produces: `PUT /api/instances/:id/name` returning `{instance_id, name, slug, login_url}`.
|
||||
|
||||
- [ ] **Step 1: Write the handler**
|
||||
|
||||
Append to `admin/internal/api/customer.go`:
|
||||
|
||||
```go
|
||||
// renameInstance changes a cloud instance's name and moves it to the slug that
|
||||
// name derives to.
|
||||
//
|
||||
// The control plane is written FIRST, because instances.slug carries the unique
|
||||
// index and that index is what actually settles a race between two accounts
|
||||
// reaching for the same name. Admin's own row follows; if that write fails the
|
||||
// control plane is put back, because HQ printing a host that is not the host is
|
||||
// worse than a failed rename.
|
||||
//
|
||||
// No licence is issued and Paddle is not called: a licence binds the instance
|
||||
// UUID, and a rename does not change it.
|
||||
func renameInstance(c *gin.Context) {
|
||||
inst, ok := ownedInstance(c, c.Param("id"))
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if inst.Deployment != license.DeploymentCloud {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": selfHostedRefusal})
|
||||
return
|
||||
}
|
||||
if inst.Placeholder {
|
||||
c.JSON(http.StatusConflict, gin.H{"error": "this instance is not provisioned yet"})
|
||||
return
|
||||
}
|
||||
|
||||
var body struct {
|
||||
Name string `json:"name"`
|
||||
}
|
||||
if err := c.ShouldBindJSON(&body); err != nil {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": "name is required"})
|
||||
return
|
||||
}
|
||||
name := strings.TrimSpace(body.Name)
|
||||
if name == "" {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": "name is required"})
|
||||
return
|
||||
}
|
||||
|
||||
if inst.RenamedAt != nil {
|
||||
if until := inst.RenamedAt.Add(models.RenameCooldown); time.Now().UTC().Before(until) {
|
||||
c.JSON(http.StatusTooManyRequests, gin.H{
|
||||
"error": fmt.Sprintf("this instance was renamed recently; it can be renamed again after %s UTC", until.Format("2 Jan 2006 15:04")),
|
||||
"retry_after": until,
|
||||
})
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
ctx := c.Request.Context()
|
||||
renamed, err := cloudprov.RenameInstance(ctx, inst.InstanceID, name)
|
||||
switch {
|
||||
case errors.Is(err, provision.ErrSlugTaken):
|
||||
c.JSON(http.StatusConflict, gin.H{"error": "that name is already in use — try another"})
|
||||
return
|
||||
case errors.Is(err, provision.ErrNameRejected):
|
||||
c.JSON(http.StatusUnprocessableEntity, gin.H{"error": err.Error()})
|
||||
return
|
||||
case err != nil:
|
||||
log.Printf("renameInstance: control plane rename of %s: %v", inst.InstanceID, err)
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": "could not rename the instance"})
|
||||
return
|
||||
}
|
||||
|
||||
if _, err := db.Admin("admin_instances").UpdateOne(ctx,
|
||||
bson.M{"instance_id": inst.InstanceID},
|
||||
bson.M{"$set": bson.M{
|
||||
"name": renamed.Name,
|
||||
"slug": renamed.Slug,
|
||||
"renamed_at": time.Now().UTC(),
|
||||
}}); err != nil {
|
||||
if rbErr := cloudprov.RestoreInstanceIdentity(ctx, inst.InstanceID, inst.Name, inst.Slug); rbErr != nil {
|
||||
log.Printf("renameInstance: rollback of %s failed: %v", inst.InstanceID, rbErr)
|
||||
}
|
||||
log.Printf("renameInstance: record rename of %s: %v", inst.InstanceID, err)
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": "could not rename the instance"})
|
||||
return
|
||||
}
|
||||
|
||||
s := auth.Current(c)
|
||||
audit.Write(ctx, models.AuditEntry{
|
||||
Actor: s.Email, Action: "instance.renamed", AccountID: s.AccountID,
|
||||
Target: inst.InstanceID, Detail: inst.Slug + " -> " + renamed.Slug, IP: c.ClientIP()})
|
||||
|
||||
c.JSON(http.StatusOK, gin.H{
|
||||
"instance_id": inst.InstanceID,
|
||||
"name": renamed.Name,
|
||||
"slug": renamed.Slug,
|
||||
// The same builder the licence emails use, rather than a second opinion
|
||||
// about how a tenant host is spelled. Empty when APP_LOGIN_URL is unset.
|
||||
"login_url": loginURLFor(renamed.Slug),
|
||||
})
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Check the imports**
|
||||
|
||||
`customer.go` must import `errors`, `fmt`, `log`, `net/http`, `strings`, `time`, `audit`, `auth`, `cloudprov`, `db`, `models`, `license`, `provision`, `gin`, `bson`. Most are already there — add only what the compiler asks for. `provision` is `gitea.hostxtra.co.uk/mrhid6/vantage/shared/provision`; `license` is `gitea.hostxtra.co.uk/mrhid6/vantage/shared/license`.
|
||||
|
||||
- [ ] **Step 3: Mount the route**
|
||||
|
||||
In `admin/internal/api/routes.go`, in the `cust` group beside the other instance routes (after `cust.POST("/instances/:id/claim-free", …)`):
|
||||
|
||||
```go
|
||||
// Renaming moves the instance's DNS host, so it is owner-or-admin like
|
||||
// every other instance mutation. Cloud only; the handler refuses the rest.
|
||||
cust.PUT("/instances/:id/name",
|
||||
auth.RequireAccountRole(models.AccountRoleOwner, models.AccountRoleAdmin),
|
||||
renameInstance)
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Build**
|
||||
|
||||
Run: `cd /go-projects/vantage && go build ./admin/... && go vet ./admin/internal/api/`
|
||||
Expected: clean.
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add admin/internal/api/customer.go admin/internal/api/routes.go
|
||||
git commit -m "feat: Add customer instance rename endpoint"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 4: Staff rename endpoint
|
||||
|
||||
**Files:**
|
||||
- Modify: `admin/internal/api/staff.go`
|
||||
- Modify: `admin/internal/api/routes.go` (the `staff` group, beside `staff.POST("/instances/:id/relink", …)`)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: everything Task 3 consumes, plus `db.Admin`.
|
||||
- Produces: `PUT /api/staff/instances/:id/name` returning `{instance_id, name, slug}`.
|
||||
|
||||
- [ ] **Step 1: Write the handler**
|
||||
|
||||
Append to `admin/internal/api/staff.go`:
|
||||
|
||||
```go
|
||||
// staffRenameInstance renames any instance, with no cooldown.
|
||||
//
|
||||
// It does NOT write renamed_at: a staff rename must not start the customer's
|
||||
// 24h clock, or fixing a name for someone locks them out of fixing it further.
|
||||
//
|
||||
// On self-hosted it changes admin's label only. There is no control-plane row to
|
||||
// write — the install is the customer's — and no slug, because self-hosted has
|
||||
// no tenant subdomain.
|
||||
func staffRenameInstance(c *gin.Context) {
|
||||
var body struct {
|
||||
Name string `json:"name"`
|
||||
}
|
||||
if err := c.ShouldBindJSON(&body); err != nil {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": "name is required"})
|
||||
return
|
||||
}
|
||||
name := strings.TrimSpace(body.Name)
|
||||
if name == "" {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": "name is required"})
|
||||
return
|
||||
}
|
||||
|
||||
ctx := c.Request.Context()
|
||||
var inst models.Instance
|
||||
if err := db.Admin("admin_instances").FindOne(ctx,
|
||||
bson.M{"instance_id": c.Param("id")}).Decode(&inst); err != nil {
|
||||
c.JSON(http.StatusNotFound, gin.H{"error": "not found"})
|
||||
return
|
||||
}
|
||||
|
||||
set := bson.M{"name": name}
|
||||
slug := inst.Slug
|
||||
|
||||
if inst.Deployment == license.DeploymentCloud && !inst.Placeholder {
|
||||
renamed, err := cloudprov.RenameInstance(ctx, inst.InstanceID, name)
|
||||
switch {
|
||||
case errors.Is(err, provision.ErrSlugTaken):
|
||||
c.JSON(http.StatusConflict, gin.H{"error": "that name is already in use"})
|
||||
return
|
||||
case errors.Is(err, provision.ErrNameRejected):
|
||||
c.JSON(http.StatusUnprocessableEntity, gin.H{"error": err.Error()})
|
||||
return
|
||||
case err != nil:
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
slug = renamed.Slug
|
||||
set["slug"] = renamed.Slug
|
||||
}
|
||||
|
||||
if _, err := db.Admin("admin_instances").UpdateOne(ctx,
|
||||
bson.M{"instance_id": inst.InstanceID}, bson.M{"$set": set}); err != nil {
|
||||
if inst.Deployment == license.DeploymentCloud && !inst.Placeholder {
|
||||
if rbErr := cloudprov.RestoreInstanceIdentity(ctx, inst.InstanceID, inst.Name, inst.Slug); rbErr != nil {
|
||||
log.Printf("staffRenameInstance: rollback of %s failed: %v", inst.InstanceID, rbErr)
|
||||
}
|
||||
}
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
|
||||
audit.Write(ctx, models.AuditEntry{
|
||||
Actor: auth.Current(c).Email, Action: "instance.renamed", AccountID: inst.AccountID,
|
||||
Target: inst.InstanceID, Detail: inst.Slug + " -> " + slug, IP: c.ClientIP()})
|
||||
|
||||
c.JSON(http.StatusOK, gin.H{"instance_id": inst.InstanceID, "name": name, "slug": slug})
|
||||
}
|
||||
```
|
||||
|
||||
`staff.go` will need `errors`, `log`, `cloudprov` and `provision` added to its imports; `fmt`, `net/http`, `strings`, `time`, `audit`, `auth`, `db`, `models`, `license`, `bson` are already there.
|
||||
|
||||
- [ ] **Step 2: Mount the route**
|
||||
|
||||
In `routes.go`, in the `staff` group after `staff.POST("/instances/:id/relink", staffRelink)`:
|
||||
|
||||
```go
|
||||
staff.PUT("/instances/:id/name", staffRenameInstance)
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Build**
|
||||
|
||||
Run: `cd /go-projects/vantage && go build ./admin/... && go vet ./admin/internal/api/`
|
||||
Expected: clean.
|
||||
|
||||
- [ ] **Step 4: Commit**
|
||||
|
||||
```bash
|
||||
git add admin/internal/api/staff.go admin/internal/api/routes.go
|
||||
git commit -m "feat: Add staff instance rename endpoint"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 5: `adminsite` API client and slug preview
|
||||
|
||||
**Files:**
|
||||
- Create: `adminsite/lib/slug.ts`
|
||||
- Modify: `adminsite/lib/api.ts` (the `Instance` interface ~line 123; the `api` object's instance calls ~line 305; `api.staff` ~line 360)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `PUT /api/instances/:id/name`, `PUT /api/staff/instances/:id/name` (Tasks 3 and 4).
|
||||
- Produces:
|
||||
- `INSTANCE_DOMAIN`, `slugify(name: string): string`, `slugError(name: string): string | undefined` from `@/lib/slug`
|
||||
- `RenameResult` interface, `api.renameInstance(id, name): Promise<RenameResult>`, `api.staff.renameInstance(id, name): Promise<RenameResult>`
|
||||
- `Instance.renamed_at?: string`
|
||||
|
||||
- [ ] **Step 1: Create the slug mirror**
|
||||
|
||||
Create `adminsite/lib/slug.ts`:
|
||||
|
||||
```ts
|
||||
/*
|
||||
* A TypeScript mirror of shared/provision's slug rules, used ONLY to preview the
|
||||
* host a rename would move an instance to while the customer types.
|
||||
*
|
||||
* It is a second implementation of Slugify, BaseSlug and ReservedSlugs, and it
|
||||
* must change in the same commit as the Go one — the same hazard as
|
||||
* web/lib/targets.ts. The preview is a courtesy; the server's 409 is the
|
||||
* boundary, and the two are allowed to disagree without anything breaking.
|
||||
*/
|
||||
|
||||
/** Mirrors provision.MinSlugLength / MaxSlugLength. */
|
||||
export const MIN_SLUG_LENGTH = 3;
|
||||
export const MAX_SLUG_LENGTH = 40;
|
||||
|
||||
/** Mirrors provision.ReservedSlugs. */
|
||||
const RESERVED = new Set([
|
||||
"www", "api", "app", "admin", "auth",
|
||||
"install", "static", "_next", "default",
|
||||
]);
|
||||
|
||||
/*
|
||||
* The tenant subdomain namespace. Also hardcoded in InstanceRecord.tsx and the
|
||||
* customer instance page; those predate this file and are left alone rather than
|
||||
* refactored under a rename change.
|
||||
*/
|
||||
export const INSTANCE_DOMAIN = "vantage.hostxtra.co.uk";
|
||||
|
||||
/** Mirrors provision.Slugify. */
|
||||
export function slugify(name: string): string {
|
||||
return name
|
||||
.toLowerCase()
|
||||
.replace(/[^a-z0-9]+/g, "-")
|
||||
.replace(/^-+|-+$/g, "");
|
||||
}
|
||||
|
||||
/** Mirrors provision.BaseSlug's truncation. */
|
||||
export function baseSlug(name: string): string {
|
||||
return slugify(name).slice(0, MAX_SLUG_LENGTH);
|
||||
}
|
||||
|
||||
/** The reason a name cannot become a slug, or undefined when it can. */
|
||||
export function slugError(name: string): string | undefined {
|
||||
const base = slugify(name);
|
||||
if (base.length < MIN_SLUG_LENGTH) {
|
||||
return `Needs at least ${MIN_SLUG_LENGTH} letters or digits.`;
|
||||
}
|
||||
if (RESERVED.has(base.slice(0, MAX_SLUG_LENGTH))) {
|
||||
return "That name is reserved.";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/** The host an instance on this slug is reached at. */
|
||||
export function hostFor(slug: string): string {
|
||||
return `${slug}.${INSTANCE_DOMAIN}`;
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Extend the API client**
|
||||
|
||||
In `adminsite/lib/api.ts`, add `renamed_at` to `Instance` (after `relink_count`):
|
||||
|
||||
```ts
|
||||
renamed_at?: string;
|
||||
```
|
||||
|
||||
Add the response type beside the other interfaces:
|
||||
|
||||
```ts
|
||||
export interface RenameResult {
|
||||
instance_id: string;
|
||||
name: string;
|
||||
slug: string;
|
||||
/** Empty when APP_LOGIN_URL is unset on the server. */
|
||||
login_url?: string;
|
||||
}
|
||||
```
|
||||
|
||||
Add the call to the `api` object, after `renewInstance`:
|
||||
|
||||
```ts
|
||||
renameInstance: (id: string, name: string) =>
|
||||
put<RenameResult>(`/api/instances/${id}/name`, { name }),
|
||||
```
|
||||
|
||||
And to `api.staff`, after `relink`:
|
||||
|
||||
```ts
|
||||
renameInstance: (id: string, name: string) =>
|
||||
put<RenameResult>(`/api/staff/instances/${id}/name`, { name }),
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Type-check**
|
||||
|
||||
Run: `cd /go-projects/vantage/adminsite && npx tsc --noEmit`
|
||||
Expected: no errors.
|
||||
|
||||
- [ ] **Step 4: Commit**
|
||||
|
||||
```bash
|
||||
git add adminsite/lib/slug.ts adminsite/lib/api.ts
|
||||
git commit -m "feat: Add rename calls and slug preview to the HQ client"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 6: The rename panel and the customer instance page
|
||||
|
||||
**Files:**
|
||||
- Create: `adminsite/components/RenamePanel.tsx`
|
||||
- Modify: `adminsite/app/(customer)/instances/[id]/page.tsx`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `api.renameInstance` / `api.staff.renameInstance`, `RenameResult` (Task 5); `slugError`, `baseSlug`, `hostFor` (Task 5); `Panel`, `Note` (`@/components/Panel`), `Button` (`@/components/Button`), `Field` (`@/components/Field`), `ApiError` (`@/lib/api`).
|
||||
- Produces: `RenamePanel({ currentName, currentSlug, onRename })` — a default-collapsed control; `onRename` is `(name: string) => Promise<RenameResult>`.
|
||||
|
||||
- [ ] **Step 1: Create the component**
|
||||
|
||||
Create `adminsite/components/RenamePanel.tsx`:
|
||||
|
||||
```tsx
|
||||
"use client";
|
||||
|
||||
import { useState } from "react";
|
||||
import { Button } from "./Button";
|
||||
import { Field } from "./Field";
|
||||
import { Note } from "./Panel";
|
||||
import { ApiError, type RenameResult } from "@/lib/api";
|
||||
import { baseSlug, hostFor, slugError } from "@/lib/slug";
|
||||
|
||||
/*
|
||||
* The rename control, and only the control — the same shape as RelinkPanel: an
|
||||
* input that expands in place rather than a modal, because this app has no modal
|
||||
* and one action with one field does not need one.
|
||||
*
|
||||
* The host preview is drawn from lib/slug.ts, a mirror of the Go rules. It can
|
||||
* disagree with the server; the 409 that comes back is the answer that counts.
|
||||
*/
|
||||
export function RenamePanel({
|
||||
currentName,
|
||||
currentSlug,
|
||||
onRename,
|
||||
}: {
|
||||
currentName: string;
|
||||
currentSlug: string;
|
||||
onRename: (name: string) => Promise<RenameResult>;
|
||||
}) {
|
||||
const [open, setOpen] = useState(false);
|
||||
const [value, setValue] = useState(currentName);
|
||||
const [error, setError] = useState<string | undefined>();
|
||||
const [busy, setBusy] = useState(false);
|
||||
const [done, setDone] = useState<RenameResult | undefined>();
|
||||
|
||||
const name = value.trim();
|
||||
const derived = baseSlug(name);
|
||||
const invalid = slugError(name);
|
||||
// A cosmetic edit that lands on the same slug is still a rename worth doing —
|
||||
// the name is what the customer reads. Only an empty or unchanged name is
|
||||
// nothing to submit.
|
||||
const unchanged = name === currentName.trim();
|
||||
|
||||
async function submit() {
|
||||
setError(undefined);
|
||||
setBusy(true);
|
||||
try {
|
||||
const res = await onRename(name);
|
||||
setDone(res);
|
||||
setOpen(false);
|
||||
} catch (err) {
|
||||
setError(err instanceof ApiError ? err.message : "Rename failed. Try again.");
|
||||
} finally {
|
||||
setBusy(false);
|
||||
}
|
||||
}
|
||||
|
||||
if (done) {
|
||||
const host = done.login_url || `https://${hostFor(done.slug)}`;
|
||||
return (
|
||||
<Note tone="warn">
|
||||
<span className="grid gap-2">
|
||||
<span>
|
||||
This instance is now <strong>{done.name}</strong>, at{" "}
|
||||
<span className="font-mono">{hostFor(done.slug)}</span>. The old address has stopped working, and your
|
||||
sign-in does not follow it — you will need to sign in again there.
|
||||
</span>
|
||||
<a href={host} className="justify-self-start font-mono text-[0.78rem] text-accent underline">
|
||||
Open {hostFor(done.slug)} →
|
||||
</a>
|
||||
</span>
|
||||
</Note>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="grid gap-3">
|
||||
{open && (
|
||||
<Field
|
||||
label="Instance name"
|
||||
value={value}
|
||||
onChange={(e) => setValue(e.target.value)}
|
||||
error={error ?? (name ? invalid : undefined)}
|
||||
hint={
|
||||
name && !invalid ? (
|
||||
<>
|
||||
Moves to <span className="font-mono">{hostFor(derived)}</span>
|
||||
{derived === currentSlug && " — the address does not change"}
|
||||
</>
|
||||
) : (
|
||||
"Letters and digits; everything else becomes a hyphen."
|
||||
)
|
||||
}
|
||||
/>
|
||||
)}
|
||||
<div className="flex flex-wrap items-center gap-3">
|
||||
<Button
|
||||
type="button"
|
||||
variant="line"
|
||||
disabled={busy || (open && (!name || Boolean(invalid) || unchanged))}
|
||||
onClick={() => (open ? submit() : setOpen(true))}
|
||||
>
|
||||
{busy ? "Renaming…" : "Rename instance"}
|
||||
</Button>
|
||||
{open && (
|
||||
<span className="text-[0.82rem] text-ink-3">
|
||||
Anyone signed in will need to sign in again at the new address, and links to the old one stop working.
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
```
|
||||
|
||||
`Note` is `({ tone = "accent" | "warn" | "expired", children })` and renders a `<p>`, which is why the success state wraps its two lines in a `<span className="grid gap-2">` rather than block elements.
|
||||
|
||||
- [ ] **Step 2: Mount it on the customer instance page**
|
||||
|
||||
In `adminsite/app/(customer)/instances/[id]/page.tsx`:
|
||||
|
||||
Add the imports:
|
||||
|
||||
```tsx
|
||||
import { RenamePanel } from "@/components/RenamePanel";
|
||||
```
|
||||
|
||||
and
|
||||
|
||||
```tsx
|
||||
import { useSession } from "@/lib/session";
|
||||
```
|
||||
|
||||
Inside `InstancePage`, with the other hooks (hooks must precede the early returns already in this component):
|
||||
|
||||
```tsx
|
||||
// useSession is the app's one way to ask who the caller is — it shares the
|
||||
// ["me"] query, so this adds no request.
|
||||
const { session } = useSession();
|
||||
```
|
||||
|
||||
and after the `cloud` const:
|
||||
|
||||
```tsx
|
||||
const mayRename = session?.account_role === "owner" || session?.account_role === "admin";
|
||||
```
|
||||
|
||||
Then add the panel to `PageFrame`'s children, directly after the `MembersPanel` line:
|
||||
|
||||
```tsx
|
||||
{/*
|
||||
* Address rather than "Rename": the panel is about where this
|
||||
* instance lives, and the rename is how you change it. Cloud
|
||||
* only — a self-hosted install has no tenant subdomain for us to
|
||||
* move.
|
||||
*/}
|
||||
{cloud && mayRename && (
|
||||
<Panel title="Address" meta={host ?? undefined}>
|
||||
<p className="text-[0.86rem] text-ink-2">
|
||||
The instance name is where its address comes from. Renaming moves it to a new address and releases the old
|
||||
one, so saved links and bookmarks to it stop working.
|
||||
</p>
|
||||
<RenamePanel
|
||||
currentName={instance.name}
|
||||
currentSlug={instance.slug ?? ""}
|
||||
onRename={async (name) => {
|
||||
const res = await api.renameInstance(instance.instance_id, name);
|
||||
qc.invalidateQueries({ queryKey: ["account"] });
|
||||
return res;
|
||||
}}
|
||||
/>
|
||||
</Panel>
|
||||
)}
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Build**
|
||||
|
||||
Run: `cd /go-projects/vantage/adminsite && npm run build`
|
||||
Expected: build succeeds.
|
||||
|
||||
- [ ] **Step 4: Commit**
|
||||
|
||||
```bash
|
||||
git add adminsite/components/RenamePanel.tsx "adminsite/app/(customer)/instances/[id]/page.tsx"
|
||||
git commit -m "feat: Let a customer rename a cloud instance from HQ"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 7: Staff instance page rename
|
||||
|
||||
**Files:**
|
||||
- Modify: `adminsite/app/(staff)/staff/instances/[id]/page.tsx`
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: `RenamePanel` (Task 6), `api.staff.renameInstance` (Task 5).
|
||||
- Produces: nothing later tasks depend on.
|
||||
|
||||
- [ ] **Step 1: Add the panel**
|
||||
|
||||
In `adminsite/app/(staff)/staff/instances/[id]/page.tsx`, add the imports:
|
||||
|
||||
```tsx
|
||||
import { RenamePanel } from "@/components/RenamePanel";
|
||||
```
|
||||
|
||||
and, inside `StaffInstancePage`, add `const qc = useQueryClient();` at the top of the component if it is not already there (`useQueryClient` is already imported for `EntitlementSection`).
|
||||
|
||||
Add this panel after the "Licence history" panel:
|
||||
|
||||
```tsx
|
||||
{/*
|
||||
* Staff rename has no cooldown and does not start the customer's:
|
||||
* fixing a name on someone's behalf must not spend their next 24
|
||||
* hours.
|
||||
*/}
|
||||
<Panel title="Name" meta={data.instance.deployment === "cloud" ? "Moves the address" : "Label only"}>
|
||||
<RenamePanel
|
||||
currentName={data.instance.name}
|
||||
currentSlug={data.instance.slug ?? ""}
|
||||
onRename={async (name) => {
|
||||
const res = await api.staff.renameInstance(data.instance.instance_id, name);
|
||||
qc.invalidateQueries({ queryKey: ["staff-instance", id] });
|
||||
return res;
|
||||
}}
|
||||
/>
|
||||
</Panel>
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Build**
|
||||
|
||||
Run: `cd /go-projects/vantage/adminsite && npm run build`
|
||||
Expected: build succeeds.
|
||||
|
||||
- [ ] **Step 3: Commit**
|
||||
|
||||
```bash
|
||||
git add "adminsite/app/(staff)/staff/instances/[id]/page.tsx"
|
||||
git commit -m "feat: Let staff rename an instance"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 8: Documentation and end-to-end verification
|
||||
|
||||
**Files:**
|
||||
- Modify: `CLAUDE.md` (the Admin REST API route list, and the `admin_instances` note under MongoDB Collections)
|
||||
|
||||
**Interfaces:**
|
||||
- Consumes: everything above.
|
||||
- Produces: nothing.
|
||||
|
||||
- [ ] **Step 1: Update the Admin REST API route list**
|
||||
|
||||
In `CLAUDE.md`, in the customer-session block, after the `POST /instances/:id/claim-free` line:
|
||||
|
||||
```
|
||||
PUT /instances/:id/name # rename a cloud instance; moves its slug (owner|admin, 24h cooldown)
|
||||
```
|
||||
|
||||
and in the staff-session block, after `POST /instances/:id/issue · /instances/:id/relink`:
|
||||
|
||||
```
|
||||
PUT /instances/:id/name # rename any instance, no cooldown
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Add the design note**
|
||||
|
||||
In `CLAUDE.md`, under "Grants project, they do not federate" (admin's control-plane write boundary is described nearby), add a short paragraph:
|
||||
|
||||
```markdown
|
||||
**A rename moves the host, and the licence does not care.** `PUT
|
||||
/api/instances/:id/name` re-derives the slug from the new name through
|
||||
`provision.RenameSlug` — the same rules that named the instance at creation —
|
||||
and writes the control plane first, because `instances.slug`'s unique index is
|
||||
what settles a race between two accounts reaching for one name. A taken slug is
|
||||
a refusal, not an `acme-2`: creation appends a counter because any free slug
|
||||
will do, and a rename is a request for one specific host. A licence binds the
|
||||
instance UUID, so nothing is reissued and Paddle is not called. The old host
|
||||
keeps resolving for up to 60s (`instancehost.go`'s cache, which admin cannot
|
||||
reach into), and `km_session` is host-only, so the customer signs in again on
|
||||
the new address — the portal says so rather than redirecting them into a login
|
||||
screen with no explanation. The 24h cooldown lives on `admin_instances.renamed_at`
|
||||
because it is admin's policy; staff bypass it and must not write the field.
|
||||
```
|
||||
|
||||
- [ ] **Step 3: Full build**
|
||||
|
||||
Run:
|
||||
```bash
|
||||
cd /go-projects/vantage && go build ./... && go vet ./admin/... ./shared/... && (cd adminsite && npm run build)
|
||||
```
|
||||
Expected: all clean.
|
||||
|
||||
- [ ] **Step 4: Manual verification against a running stack**
|
||||
|
||||
Work through each and record the result:
|
||||
|
||||
1. Rename a cloud instance from `/instances/<id>` as an owner. Panel reports the new host.
|
||||
2. In Mongo: `db.instances.findOne({instance_id})` and `db.admin_instances.findOne({instance_id})` agree on `name` and `slug`; `admin_instances.renamed_at` is set.
|
||||
3. The new host serves a login page. The old host stops resolving to the instance within ~60 seconds.
|
||||
4. A second rename inside 24 hours answers `429` with the unlock time.
|
||||
5. Renaming onto a slug another instance holds answers `409` and changes neither database.
|
||||
6. `PUT /api/instances/:id/name` on a self-hosted instance answers `400` with the `selfHostedRefusal` message.
|
||||
7. `GET /api/staff/audit` shows `instance.renamed` with `old-slug -> new-slug`.
|
||||
8. Staff rename of the same instance succeeds immediately and leaves `renamed_at` unchanged.
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
git add CLAUDE.md
|
||||
git commit -m "docs: Document instance rename in HQ"
|
||||
```
|
||||
|
||||
- [ ] **Step 6: Refresh the knowledge graph**
|
||||
|
||||
```bash
|
||||
graphify update .
|
||||
```
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,641 @@
|
||||
<title>Vantage Status Pages</title>
|
||||
<style>
|
||||
:root{
|
||||
color-scheme: dark;
|
||||
/* Vantage web/ dark tokens, copied verbatim from web/app/globals.css.
|
||||
This mockup commits to one theme because web/ does. */
|
||||
--ground:#071628; --panel:#0d2138; --panel-2:#102842; --well:#04101f;
|
||||
--ink:#e4ecf6; --ink-2:#9fb3ca; --ink-3:#71879f;
|
||||
--rule:#1e3855; --rule-soft:#172c44;
|
||||
--accent:#5b9be8; --accent-hover:#7fb2f0; --accent-ink:#04101f;
|
||||
--up:#4fb484; --pend:#d6a63f; --down:#e2705a; --logo:#7fb2f0;
|
||||
--shadow:0 1px 0 rgba(0,0,0,.35), 0 20px 44px -26px rgba(0,0,0,.85);
|
||||
--sans: ui-sans-serif, system-ui, -apple-system, "Segoe UI", Roboto, "Helvetica Neue", Arial, sans-serif;
|
||||
--mono: ui-monospace, "Cascadia Mono", "SF Mono", "JetBrains Mono", Menlo, Consolas, monospace;
|
||||
--r:4px;
|
||||
}
|
||||
*{box-sizing:border-box;margin:0;padding:0}
|
||||
body{background:var(--ground);color:var(--ink);font-family:var(--sans);-webkit-font-smoothing:antialiased;line-height:1.5}
|
||||
a{color:inherit}
|
||||
:focus-visible{outline:2px solid var(--accent);outline-offset:2px}
|
||||
|
||||
.page{max-width:1180px;margin:0 auto;padding:48px 24px 96px;display:flex;flex-direction:column;gap:56px}
|
||||
|
||||
.lede h1{font-size:1.6rem;font-weight:800;letter-spacing:-.035em;text-wrap:balance}
|
||||
.lede p{color:var(--ink-2);font-size:.9rem;max-width:65ch;margin-top:8px}
|
||||
|
||||
.cap{font-family:var(--mono);font-size:.68rem;text-transform:uppercase;letter-spacing:.1em;color:var(--ink-2)}
|
||||
|
||||
.board{display:flex;flex-direction:column;gap:10px}
|
||||
.board__head{display:flex;align-items:baseline;justify-content:space-between;gap:16px;flex-wrap:wrap}
|
||||
.board__route{font-family:var(--mono);font-size:.7rem;color:var(--ink-3)}
|
||||
.frame{border:1px solid var(--rule);border-radius:var(--r);background:var(--ground);box-shadow:var(--shadow);overflow:hidden}
|
||||
|
||||
/* address strip — shows the URL scheme being approved */
|
||||
.addr{display:flex;align-items:center;gap:10px;background:var(--well);border-bottom:1px solid var(--rule);padding:9px 14px}
|
||||
.addr__dots{display:flex;gap:5px}
|
||||
.addr__dots i{width:8px;height:8px;border-radius:999px;background:var(--rule);display:block}
|
||||
.addr__url{font-family:var(--mono);font-size:.72rem;color:var(--ink-2);overflow-x:auto;white-space:nowrap}
|
||||
.addr__url b{color:var(--ink);font-weight:600}
|
||||
.addr__tag{margin-left:auto;font-family:var(--mono);font-size:.62rem;text-transform:uppercase;letter-spacing:.1em;color:var(--ink-3);border:1px solid var(--rule);border-radius:999px;padding:2px 8px;white-space:nowrap}
|
||||
|
||||
/* ---------- public status page ---------- */
|
||||
.pub{padding:40px 28px 32px}
|
||||
.pub__inner{max-width:720px;margin:0 auto;display:flex;flex-direction:column;gap:28px}
|
||||
.pub__head{display:flex;align-items:center;gap:14px}
|
||||
.mark{width:38px;height:38px;border-radius:var(--r);background:var(--panel-2);border:1px solid var(--rule);display:grid;place-items:center;color:var(--logo);font-family:var(--mono);font-weight:700;font-size:.85rem;flex-shrink:0}
|
||||
.pub__head h2{font-size:1.35rem;font-weight:800;letter-spacing:-.03em}
|
||||
.pub__head p{color:var(--ink-2);font-size:.85rem;margin-top:2px}
|
||||
|
||||
.overall{display:flex;align-items:center;gap:11px;border:1px solid;border-radius:var(--r);padding:14px 16px;font-weight:600;font-size:.95rem}
|
||||
.overall--down{background:rgba(226,112,90,.10);border-color:rgba(226,112,90,.30);color:var(--down)}
|
||||
.glyph{width:16px;height:16px;flex-shrink:0}
|
||||
|
||||
.banner{border:1px solid var(--rule);background:var(--panel);border-radius:var(--r);padding:12px 14px;font-size:.85rem;color:var(--ink-2);display:flex;gap:10px}
|
||||
.banner b{color:var(--ink);font-weight:600}
|
||||
|
||||
.group{display:flex;flex-direction:column;gap:10px}
|
||||
.group > .cap{padding-left:2px}
|
||||
|
||||
.card{border:1px solid var(--rule);background:var(--panel);border-radius:var(--r)}
|
||||
.rows > * + *{border-top:1px solid var(--rule-soft)}
|
||||
|
||||
.comp{padding:16px}
|
||||
.comp__top{display:flex;align-items:center;justify-content:space-between;gap:16px;margin-bottom:10px}
|
||||
.comp__name{font-weight:600;font-size:.92rem}
|
||||
.state{display:inline-flex;align-items:center;gap:7px;font-size:.78rem;color:var(--ink-2);white-space:nowrap}
|
||||
.dot{width:7px;height:7px;border-radius:999px;display:block;flex-shrink:0}
|
||||
.dot--up{background:var(--up)} .dot--down{background:var(--down)}
|
||||
.dot--maint{background:var(--accent)} .dot--pend{background:var(--pend)}
|
||||
.dot--none{background:var(--rule)}
|
||||
|
||||
.bar{display:flex;gap:2px;overflow-x:auto;padding-bottom:2px}
|
||||
.bar span{height:26px;width:3px;border-radius:999px;flex:0 0 auto;background:var(--rule)}
|
||||
.bar .up{background:var(--up)} .bar .down{background:var(--down)}
|
||||
.bar .maint{background:var(--accent)} .bar .none{background:var(--rule)}
|
||||
.scale{display:flex;justify-content:space-between;margin-top:7px;font-size:.7rem;color:var(--ink-3)}
|
||||
.scale b{color:var(--ink-2);font-weight:600;font-variant-numeric:tabular-nums}
|
||||
|
||||
.inc{padding:14px 16px}
|
||||
.inc__top{display:flex;align-items:baseline;justify-content:space-between;gap:14px}
|
||||
.inc__title{font-weight:600;font-size:.92rem}
|
||||
.inc__meta{font-size:.75rem;color:var(--ink-3);margin-top:3px}
|
||||
.inc__affects{font-size:.75rem;color:var(--ink-2);margin-top:5px}
|
||||
.pill{font-family:var(--mono);font-size:.62rem;text-transform:uppercase;letter-spacing:.1em;border-radius:999px;padding:3px 9px;border:1px solid;white-space:nowrap}
|
||||
.pill--inv{color:var(--down);border-color:rgba(226,112,90,.35);background:rgba(226,112,90,.10)}
|
||||
.pill--mon{color:var(--pend);border-color:rgba(214,166,63,.35);background:rgba(214,166,63,.10)}
|
||||
.pill--res{color:var(--up);border-color:rgba(79,180,132,.35);background:rgba(79,180,132,.10)}
|
||||
.pill--sch{color:var(--accent);border-color:rgba(91,155,232,.35);background:rgba(91,155,232,.10)}
|
||||
.pill--draft{color:var(--ink-2);border-color:var(--rule);background:var(--panel-2)}
|
||||
.pill--live{color:var(--up);border-color:rgba(79,180,132,.35);background:rgba(79,180,132,.10)}
|
||||
|
||||
.timeline{margin-top:12px;border-left:1px solid var(--rule);padding-left:14px;display:flex;flex-direction:column;gap:12px}
|
||||
.tl__head{display:flex;align-items:baseline;gap:9px}
|
||||
.tl__st{font-family:var(--mono);font-size:.62rem;text-transform:uppercase;letter-spacing:.1em;color:var(--ink-2)}
|
||||
.tl__at{font-size:.7rem;color:var(--ink-3);font-variant-numeric:tabular-nums}
|
||||
.tl__body{font-size:.85rem;margin-top:3px;color:var(--ink)}
|
||||
|
||||
.pub__foot{text-align:center;font-size:.72rem;color:var(--ink-3);padding-top:6px}
|
||||
|
||||
/* ---------- editor ---------- */
|
||||
.app{display:grid;grid-template-columns:236px 1fr;min-height:660px}
|
||||
.side{background:var(--panel);border-right:1px solid var(--rule);display:flex;flex-direction:column}
|
||||
.side__brand{height:64px;display:flex;align-items:center;gap:12px;padding:0 20px;border-bottom:1px solid var(--rule);flex-shrink:0}
|
||||
.side__brand .mark{width:32px;height:32px;font-size:.78rem}
|
||||
.side__brand b{font-size:1rem;font-weight:800;letter-spacing:-.035em;display:block;line-height:1.2}
|
||||
.side__nav{padding:16px 12px;display:flex;flex-direction:column;gap:16px}
|
||||
.navgrp + .navgrp{border-top:1px solid var(--rule);padding-top:16px}
|
||||
.navgrp > .cap{padding:0 12px 6px}
|
||||
.navgrp ul{list-style:none;display:flex;flex-direction:column;gap:4px}
|
||||
.navgrp a{position:relative;display:flex;align-items:center;gap:12px;border-radius:var(--r);padding:9px 12px;font-size:.85rem;font-weight:500;color:var(--ink-2);text-decoration:none}
|
||||
.navgrp a:hover{background:var(--panel-2);color:var(--ink)}
|
||||
.navgrp a.on{background:var(--panel-2);color:var(--ink);font-weight:600}
|
||||
.navgrp a.on::before{content:"";position:absolute;left:0;top:4px;bottom:4px;width:2px;border-radius:999px;background:var(--accent)}
|
||||
.navgrp svg{width:16px;height:16px;flex-shrink:0;opacity:.9}
|
||||
|
||||
.main{padding:26px 28px 36px;display:flex;flex-direction:column;gap:22px;min-width:0}
|
||||
.back{font-size:.78rem;color:var(--ink-2);text-decoration:none;display:inline-flex;gap:6px;align-items:center}
|
||||
.back:hover{color:var(--ink)}
|
||||
.phead{display:flex;align-items:flex-start;justify-content:space-between;gap:20px;flex-wrap:wrap}
|
||||
.phead h2{font-size:1.3rem;font-weight:800;letter-spacing:-.03em}
|
||||
.record{display:flex;align-items:center;gap:8px;margin-top:6px}
|
||||
.record code{font-family:var(--mono);font-size:.72rem;color:var(--ink-2);background:var(--well);border:1px solid var(--rule);border-radius:var(--r);padding:3px 8px}
|
||||
.copy{background:none;border:0;color:var(--ink-3);cursor:pointer;font-size:.72rem;font-family:var(--mono)}
|
||||
.copy:hover{color:var(--accent)}
|
||||
.acts{display:flex;gap:9px;flex-wrap:wrap}
|
||||
.btn{font-size:.82rem;font-weight:600;border-radius:var(--r);padding:8px 14px;border:1px solid var(--rule);background:var(--panel);color:var(--ink);cursor:pointer;text-decoration:none;display:inline-flex;align-items:center;gap:7px}
|
||||
.btn:hover{background:var(--panel-2)}
|
||||
.btn--p{background:var(--accent);border-color:var(--accent);color:var(--accent-ink)}
|
||||
.btn--p:hover{background:var(--accent-hover)}
|
||||
|
||||
.panel{border:1px solid var(--rule);background:var(--panel);border-radius:var(--r)}
|
||||
.panel__head{display:flex;align-items:center;justify-content:space-between;gap:14px;padding:13px 16px;border-bottom:1px solid var(--rule)}
|
||||
.panel__head h3{font-size:.95rem;font-weight:700}
|
||||
.panel__head p{font-size:.76rem;color:var(--ink-3);margin-top:2px}
|
||||
.panel__body{padding:16px;display:flex;flex-direction:column;gap:16px}
|
||||
|
||||
.fields{display:grid;grid-template-columns:repeat(auto-fit,minmax(230px,1fr));gap:14px}
|
||||
.field{display:flex;flex-direction:column;gap:6px;min-width:0}
|
||||
.field > label{font-size:.76rem;font-weight:600;color:var(--ink-2)}
|
||||
.field .hint{font-size:.72rem;color:var(--ink-3)}
|
||||
.in{background:var(--well);border:1px solid var(--rule);border-radius:var(--r);padding:8px 11px;font:inherit;font-size:.85rem;color:var(--ink);width:100%}
|
||||
.in::placeholder{color:var(--ink-3)}
|
||||
.in:focus{outline:2px solid var(--accent);outline-offset:-1px;border-color:var(--accent)}
|
||||
.in--mono{font-family:var(--mono);font-size:.8rem}
|
||||
|
||||
.toggle{display:flex;align-items:center;justify-content:space-between;gap:16px;background:var(--panel-2);border:1px solid var(--rule);border-radius:var(--r);padding:12px 14px}
|
||||
.toggle p{font-size:.76rem;color:var(--ink-3);margin-top:3px;max-width:52ch}
|
||||
.toggle b{font-size:.85rem}
|
||||
.sw{width:38px;height:21px;border-radius:999px;background:var(--up);border:0;position:relative;cursor:pointer;flex-shrink:0}
|
||||
.sw::after{content:"";position:absolute;top:2px;left:19px;width:17px;height:17px;border-radius:999px;background:var(--accent-ink)}
|
||||
.sw[aria-checked="false"]{background:var(--rule)}
|
||||
.sw[aria-checked="false"]::after{left:2px;background:var(--ink-3)}
|
||||
|
||||
.sect{border:1px solid var(--rule);border-radius:var(--r);background:var(--panel-2)}
|
||||
.sect__head{display:flex;align-items:center;gap:10px;padding:10px 12px;border-bottom:1px solid var(--rule)}
|
||||
.sect__head .in{max-width:220px}
|
||||
.sect__head .rm{margin-left:auto}
|
||||
.rm{background:none;border:0;color:var(--ink-3);font-size:.75rem;cursor:pointer;font-family:var(--mono)}
|
||||
.rm:hover{color:var(--down)}
|
||||
.entry{display:grid;grid-template-columns:1fr 1fr auto;gap:12px;align-items:center;padding:11px 12px}
|
||||
.entry + .entry{border-top:1px solid var(--rule-soft)}
|
||||
.entry__mon{display:flex;flex-direction:column;gap:2px;min-width:0}
|
||||
.entry__mon b{font-size:.84rem;font-weight:600}
|
||||
.entry__mon span{font-family:var(--mono);font-size:.68rem;color:var(--ink-3)}
|
||||
.adds{display:flex;gap:9px;flex-wrap:wrap;padding:0 12px 12px}
|
||||
|
||||
.inc-row{display:flex;align-items:flex-start;justify-content:space-between;gap:14px;padding:13px 14px}
|
||||
.inc-row + .inc-row{border-top:1px solid var(--rule-soft)}
|
||||
.inc-row__l{min-width:0}
|
||||
.inc-row__l b{font-size:.88rem;font-weight:600;display:block}
|
||||
.inc-row__l span{font-size:.74rem;color:var(--ink-3)}
|
||||
.inc-row__r{display:flex;align-items:center;gap:9px;flex-shrink:0}
|
||||
|
||||
.notes{border-top:1px solid var(--rule);padding-top:14px;display:flex;flex-direction:column;gap:7px}
|
||||
.notes li{font-size:.82rem;color:var(--ink-2);display:flex;gap:10px;list-style:none}
|
||||
.notes li b{color:var(--ink);font-weight:600}
|
||||
.notes .k{font-family:var(--mono);font-size:.66rem;text-transform:uppercase;letter-spacing:.1em;color:var(--ink-3);flex:0 0 76px;padding-top:2px}
|
||||
|
||||
@media (max-width:820px){
|
||||
.app{grid-template-columns:1fr}
|
||||
.side{display:none}
|
||||
.entry{grid-template-columns:1fr}
|
||||
.page{padding:32px 16px 64px}
|
||||
}
|
||||
</style>
|
||||
|
||||
<div class="page">
|
||||
|
||||
<header class="lede">
|
||||
<p class="cap" style="margin-bottom:10px">Vantage · status pages · mockup for approval</p>
|
||||
<h1>Two screens: what the public sees, and what the operator edits</h1>
|
||||
<p>Drawn with the real <code style="font-family:var(--mono);font-size:.85em">web/</code> dark tokens and the existing sidebar idioms, so what gets approved here is what gets built. The public page is shown mid-incident rather than all-green, because that is the state it exists for.</p>
|
||||
</header>
|
||||
|
||||
<!-- ================= PUBLIC ================= -->
|
||||
<section class="board">
|
||||
<div class="board__head">
|
||||
<p class="cap">1 · Public status page</p>
|
||||
<p class="board__route">web/app/status/[pageId]/page.tsx · no auth, no sidebar</p>
|
||||
</div>
|
||||
|
||||
<div class="frame">
|
||||
<div class="addr">
|
||||
<span class="addr__dots"><i></i><i></i><i></i></span>
|
||||
<span class="addr__url">https://acme.vantage.example.com<b>/status/api</b></span>
|
||||
<span class="addr__tag">signed out</span>
|
||||
</div>
|
||||
|
||||
<div class="pub">
|
||||
<div class="pub__inner">
|
||||
|
||||
<div class="pub__head">
|
||||
<div class="mark">AC</div>
|
||||
<div>
|
||||
<h2>Acme Platform Status</h2>
|
||||
<p>Live availability for the Acme API and dashboard.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="overall overall--down">
|
||||
<svg class="glyph" viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.6" aria-hidden="true">
|
||||
<circle cx="8" cy="8" r="6.4"/><path d="M8 4.8v3.6M8 11.1h.01" stroke-linecap="round"/>
|
||||
</svg>
|
||||
Service disruption
|
||||
</div>
|
||||
|
||||
<div class="banner">
|
||||
<svg class="glyph" viewBox="0 0 16 16" fill="none" stroke="var(--accent)" stroke-width="1.6" aria-hidden="true" style="margin-top:2px">
|
||||
<circle cx="8" cy="8" r="6.4"/><path d="M8 7.4v3.8M8 5.1h.01" stroke-linecap="round"/>
|
||||
</svg>
|
||||
<span><b>Europe region only.</b> US and APAC are unaffected. Follow this page for updates.</span>
|
||||
</div>
|
||||
|
||||
<div class="group">
|
||||
<p class="cap">Active</p>
|
||||
<div class="card">
|
||||
<article class="inc">
|
||||
<div class="inc__top">
|
||||
<div>
|
||||
<p class="inc__title">Elevated error rates on database writes</p>
|
||||
<p class="inc__meta">Started 24 Aug 2026, 09:12 UTC</p>
|
||||
</div>
|
||||
<span class="pill pill--mon">monitoring</span>
|
||||
</div>
|
||||
<p class="inc__affects">Affects Primary database, Public API</p>
|
||||
<div class="timeline">
|
||||
<div>
|
||||
<div class="tl__head"><span class="tl__st">monitoring</span><span class="tl__at">11:40 UTC</span></div>
|
||||
<p class="tl__body">Failover completed. Write latency is back to normal and we are watching for recurrence before calling this resolved.</p>
|
||||
</div>
|
||||
<div>
|
||||
<div class="tl__head"><span class="tl__st">identified</span><span class="tl__at">09:48 UTC</span></div>
|
||||
<p class="tl__body">A failing disk on the primary database node is causing write timeouts. Failover to the standby node is in progress.</p>
|
||||
</div>
|
||||
<div>
|
||||
<div class="tl__head"><span class="tl__st">investigating</span><span class="tl__at">09:15 UTC</span></div>
|
||||
<p class="tl__body">We are investigating a rise in write errors affecting the API.</p>
|
||||
</div>
|
||||
</div>
|
||||
</article>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="group">
|
||||
<p class="cap">Scheduled maintenance</p>
|
||||
<div class="card">
|
||||
<article class="inc">
|
||||
<div class="inc__top">
|
||||
<div>
|
||||
<p class="inc__title">Object storage capacity upgrade</p>
|
||||
<p class="inc__meta">31 Aug 2026, 02:00 – 04:00 UTC</p>
|
||||
</div>
|
||||
<span class="pill pill--sch">scheduled</span>
|
||||
</div>
|
||||
<p class="inc__affects">Affects Object storage</p>
|
||||
</article>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="group">
|
||||
<p class="cap">API</p>
|
||||
<div class="card rows">
|
||||
<div class="comp" data-bar="api" data-state="down">
|
||||
<div class="comp__top">
|
||||
<span class="comp__name">Public API</span>
|
||||
<span class="state"><i class="dot dot--down"></i>Down</span>
|
||||
</div>
|
||||
<div class="bar"></div>
|
||||
<div class="scale"><span>90 days ago</span><span><b>99.81%</b> uptime</span><span>Today</span></div>
|
||||
</div>
|
||||
<div class="comp" data-bar="hooks" data-state="up">
|
||||
<div class="comp__top">
|
||||
<span class="comp__name">Webhook delivery</span>
|
||||
<span class="state"><i class="dot dot--up"></i>Operational</span>
|
||||
</div>
|
||||
<div class="bar"></div>
|
||||
<div class="scale"><span>90 days ago</span><span><b>99.99%</b> uptime</span><span>Today</span></div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="group">
|
||||
<p class="cap">Web</p>
|
||||
<div class="card rows">
|
||||
<div class="comp" data-bar="dash" data-state="up">
|
||||
<div class="comp__top">
|
||||
<span class="comp__name">Dashboard</span>
|
||||
<span class="state"><i class="dot dot--up"></i>Operational</span>
|
||||
</div>
|
||||
<div class="bar"></div>
|
||||
<div class="scale"><span>90 days ago</span><span><b>99.97%</b> uptime</span><span>Today</span></div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="group">
|
||||
<p class="cap">Data</p>
|
||||
<div class="card rows">
|
||||
<div class="comp" data-bar="db" data-state="down">
|
||||
<div class="comp__top">
|
||||
<span class="comp__name">Primary database</span>
|
||||
<span class="state"><i class="dot dot--down"></i>Down</span>
|
||||
</div>
|
||||
<div class="bar"></div>
|
||||
<div class="scale"><span>90 days ago</span><span><b>99.62%</b> uptime</span><span>Today</span></div>
|
||||
</div>
|
||||
<div class="comp" data-bar="obj" data-state="maint">
|
||||
<div class="comp__top">
|
||||
<span class="comp__name">Object storage</span>
|
||||
<span class="state"><i class="dot dot--maint"></i>Maintenance</span>
|
||||
</div>
|
||||
<div class="bar"></div>
|
||||
<div class="scale"><span>90 days ago</span><span><b>99.94%</b> uptime</span><span>Today</span></div>
|
||||
</div>
|
||||
<div class="comp" data-bar="new" data-state="up">
|
||||
<div class="comp__top">
|
||||
<span class="comp__name">Search index</span>
|
||||
<span class="state"><i class="dot dot--up"></i>Operational</span>
|
||||
</div>
|
||||
<div class="bar"></div>
|
||||
<div class="scale"><span>90 days ago</span><span><b>100.00%</b> uptime</span><span>Today</span></div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="group">
|
||||
<p class="cap">Past incidents</p>
|
||||
<div class="card rows">
|
||||
<article class="inc">
|
||||
<div class="inc__top">
|
||||
<div>
|
||||
<p class="inc__title">Public API unavailable</p>
|
||||
<p class="inc__meta">2 Aug 2026, 14:02 UTC — resolved 14:19 UTC</p>
|
||||
</div>
|
||||
<span class="pill pill--res">resolved</span>
|
||||
</div>
|
||||
<p class="inc__affects">Affects Public API</p>
|
||||
</article>
|
||||
<article class="inc">
|
||||
<div class="inc__top">
|
||||
<div>
|
||||
<p class="inc__title">Slow dashboard loads in Europe</p>
|
||||
<p class="inc__meta">17 Jul 2026, 08:30 UTC — resolved 10:05 UTC</p>
|
||||
</div>
|
||||
<span class="pill pill--res">resolved</span>
|
||||
</div>
|
||||
<p class="inc__affects">Affects Dashboard</p>
|
||||
</article>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="pub__foot">Updated 24 Aug 2026, 11:58 UTC · refreshes every 60 seconds</p>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<ul class="notes">
|
||||
<li><span class="k">Redacted</span><span>No target URL, host, port or failure text anywhere on this page. <b>Search index</b> shows the no-data tail as grey cells rather than claiming 100% for days before it existed.</span></li>
|
||||
<li><span class="k">Maintenance</span><span><b>Object storage</b> reads as Maintenance, not Down — but its uptime figure is untouched. The window changes how it is drawn, never what the numbers say.</span></li>
|
||||
<li><span class="k">Colour</span><span>Every state carries a word and a shape as well as a hue. The page is readable with colour vision differences and in greyscale print.</span></li>
|
||||
</ul>
|
||||
</section>
|
||||
|
||||
<!-- ================= EDITOR ================= -->
|
||||
<section class="board">
|
||||
<div class="board__head">
|
||||
<p class="cap">2 · Status page editor</p>
|
||||
<p class="board__route">web/app/(app)/status-pages/[pageId]/page.tsx · owner or admin</p>
|
||||
</div>
|
||||
|
||||
<div class="frame">
|
||||
<div class="app">
|
||||
<aside class="side">
|
||||
<div class="side__brand">
|
||||
<div class="mark">V</div>
|
||||
<div>
|
||||
<b>Vantage</b>
|
||||
<span class="cap">Acme Ltd</span>
|
||||
</div>
|
||||
</div>
|
||||
<nav class="side__nav">
|
||||
<div class="navgrp">
|
||||
<p class="cap">Fleet</p>
|
||||
<ul>
|
||||
<li><a href="#"><svg viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.5"><rect x="2" y="3" width="12" height="4" rx="1"/><rect x="2" y="9" width="12" height="4" rx="1"/></svg>Servers</a></li>
|
||||
<li><a href="#"><svg viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.5"><rect x="2.5" y="2.5" width="11" height="11" rx="1.5"/><path d="M6 6h4v4H6z"/></svg>Workloads</a></li>
|
||||
<li><a href="#"><svg viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.5" stroke-linecap="round"><path d="M1.5 8.5h3l2-4 3 7 2-3h3"/></svg>Monitors</a></li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="navgrp">
|
||||
<p class="cap">Access</p>
|
||||
<ul>
|
||||
<li><a href="#"><svg viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.5"><circle cx="5.5" cy="8" r="3"/><path d="M8.5 8h6M12 8v2.5"/></svg>SSH Keys</a></li>
|
||||
<li><a href="#"><svg viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.5"><rect x="3" y="7" width="10" height="6.5" rx="1.5"/><path d="M5.5 7V5a2.5 2.5 0 015 0v2"/></svg>Secrets</a></li>
|
||||
</ul>
|
||||
</div>
|
||||
<div class="navgrp">
|
||||
<p class="cap">Instance</p>
|
||||
<ul>
|
||||
<li><a href="#" class="on"><svg viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.5"><rect x="2" y="3" width="12" height="10" rx="1.5"/><path d="M4.5 10.5v-2M8 10.5v-4M11.5 10.5v-3" stroke-linecap="round"/></svg>Status Pages</a></li>
|
||||
<li><a href="#"><svg viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.5"><path d="M3 3h10v10H3z"/><path d="M5.5 6.5h5M5.5 9.5h3"/></svg>Audit Log</a></li>
|
||||
<li><a href="#"><svg viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.5"><circle cx="8" cy="8" r="2.2"/><path d="M8 1.8v1.6M8 12.6v1.6M14.2 8h-1.6M3.4 8H1.8"/></svg>Settings</a></li>
|
||||
</ul>
|
||||
</div>
|
||||
</nav>
|
||||
</aside>
|
||||
|
||||
<div class="main">
|
||||
<a class="back" href="#">
|
||||
<svg class="glyph" viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.6" stroke-linecap="round"><path d="M9.5 3.5L5 8l4.5 4.5"/></svg>
|
||||
All status pages
|
||||
</a>
|
||||
|
||||
<div class="phead">
|
||||
<div>
|
||||
<h2>Acme Platform Status</h2>
|
||||
<div class="record">
|
||||
<code>acme.vantage.example.com/status/api</code>
|
||||
<button class="copy" type="button">copy</button>
|
||||
</div>
|
||||
</div>
|
||||
<div class="acts">
|
||||
<a class="btn" href="#">
|
||||
<svg class="glyph" viewBox="0 0 16 16" fill="none" stroke="currentColor" stroke-width="1.6" stroke-linecap="round"><path d="M6.5 3.5h6v6M12.5 3.5L7 9"/><path d="M11 10.5v2h-8v-8h2"/></svg>
|
||||
View page
|
||||
</a>
|
||||
<button class="btn btn--p" type="button">Save changes</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="panel">
|
||||
<div class="panel__head">
|
||||
<div>
|
||||
<h3>Details</h3>
|
||||
<p>What visitors see at the top of the page.</p>
|
||||
</div>
|
||||
<span class="pill pill--live">published</span>
|
||||
</div>
|
||||
<div class="panel__body">
|
||||
<div class="toggle">
|
||||
<div>
|
||||
<b>Published</b>
|
||||
<p>Anyone with the link can read this page. Unpublished pages return not found, so you can compose before announcing.</p>
|
||||
</div>
|
||||
<button class="sw" type="button" role="switch" aria-checked="true" aria-label="Published"></button>
|
||||
</div>
|
||||
<div class="fields">
|
||||
<div class="field">
|
||||
<label for="f-title">Title</label>
|
||||
<input class="in" id="f-title" value="Acme Platform Status">
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="f-id">Page address</label>
|
||||
<input class="in in--mono" id="f-id" value="api" disabled>
|
||||
<span class="hint">Fixed once created — the link is already out there.</span>
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="f-desc">Description</label>
|
||||
<input class="in" id="f-desc" value="Live availability for the Acme API and dashboard.">
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="f-logo">Logo URL</label>
|
||||
<input class="in in--mono" id="f-logo" placeholder="https://acme.example.com/logo.svg">
|
||||
</div>
|
||||
</div>
|
||||
<div class="field">
|
||||
<label for="f-ban">Notice</label>
|
||||
<input class="in" id="f-ban" value="Europe region only. US and APAC are unaffected. Follow this page for updates.">
|
||||
<span class="hint">Shown above everything else. Clear it to remove the notice.</span>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="panel">
|
||||
<div class="panel__head">
|
||||
<div>
|
||||
<h3>Components</h3>
|
||||
<p>Monitors grouped for the public page. Grouping here is separate from the groups on Monitors.</p>
|
||||
</div>
|
||||
<button class="btn" type="button">Add section</button>
|
||||
</div>
|
||||
<div class="panel__body">
|
||||
|
||||
<div class="sect">
|
||||
<div class="sect__head">
|
||||
<input class="in" value="API" aria-label="Section name">
|
||||
<button class="rm" type="button">remove section</button>
|
||||
</div>
|
||||
<div class="entry">
|
||||
<div class="entry__mon">
|
||||
<b>prod-api-eu-health</b>
|
||||
<span>http · every 30s</span>
|
||||
</div>
|
||||
<input class="in" value="Public API" aria-label="Public name for prod-api-eu-health">
|
||||
<button class="rm" type="button">remove</button>
|
||||
</div>
|
||||
<div class="entry">
|
||||
<div class="entry__mon">
|
||||
<b>hooks-dispatch-probe</b>
|
||||
<span>http · every 60s</span>
|
||||
</div>
|
||||
<input class="in" value="Webhook delivery" aria-label="Public name for hooks-dispatch-probe">
|
||||
<button class="rm" type="button">remove</button>
|
||||
</div>
|
||||
<div class="adds"><button class="btn" type="button">Add monitor</button></div>
|
||||
</div>
|
||||
|
||||
<div class="sect">
|
||||
<div class="sect__head">
|
||||
<input class="in" value="Data" aria-label="Section name">
|
||||
<button class="rm" type="button">remove section</button>
|
||||
</div>
|
||||
<div class="entry">
|
||||
<div class="entry__mon">
|
||||
<b>pg-primary-10-0-0-5</b>
|
||||
<span>tcp · every 30s</span>
|
||||
</div>
|
||||
<input class="in" value="Primary database" aria-label="Public name for pg-primary-10-0-0-5">
|
||||
<button class="rm" type="button">remove</button>
|
||||
</div>
|
||||
<div class="entry">
|
||||
<div class="entry__mon">
|
||||
<b>minio-gw</b>
|
||||
<span>http · every 60s</span>
|
||||
</div>
|
||||
<input class="in" placeholder="minio-gw" aria-label="Public name for minio-gw">
|
||||
<button class="rm" type="button">remove</button>
|
||||
</div>
|
||||
<div class="adds"><button class="btn" type="button">Add monitor</button></div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="panel">
|
||||
<div class="panel__head">
|
||||
<div>
|
||||
<h3>Incidents</h3>
|
||||
<p>Written by you. Outages Vantage detects appear on the page automatically.</p>
|
||||
</div>
|
||||
<div class="acts">
|
||||
<button class="btn" type="button">Schedule maintenance</button>
|
||||
<button class="btn btn--p" type="button">Open incident</button>
|
||||
</div>
|
||||
</div>
|
||||
<div>
|
||||
<div class="inc-row">
|
||||
<div class="inc-row__l">
|
||||
<b>Elevated error rates on database writes</b>
|
||||
<span>Opened 09:12 UTC · 3 updates · affects Primary database, Public API</span>
|
||||
</div>
|
||||
<div class="inc-row__r">
|
||||
<span class="pill pill--mon">monitoring</span>
|
||||
<button class="btn" type="button">Post update</button>
|
||||
</div>
|
||||
</div>
|
||||
<div class="inc-row">
|
||||
<div class="inc-row__l">
|
||||
<b>Object storage capacity upgrade</b>
|
||||
<span>31 Aug, 02:00–04:00 UTC · affects Object storage</span>
|
||||
</div>
|
||||
<div class="inc-row__r">
|
||||
<span class="pill pill--sch">scheduled</span>
|
||||
<button class="btn" type="button">Edit</button>
|
||||
</div>
|
||||
</div>
|
||||
<div class="inc-row">
|
||||
<div class="inc-row__l">
|
||||
<b>Slow dashboard loads in Europe</b>
|
||||
<span>17 Jul · resolved after 1h 35m · affects Dashboard</span>
|
||||
</div>
|
||||
<div class="inc-row__r">
|
||||
<span class="pill pill--res">resolved</span>
|
||||
<button class="btn" type="button">Edit</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<ul class="notes">
|
||||
<li><span class="k">Naming</span><span>The monitor's own identifier stays visible on the left; the <b>public name</b> is a separate field beside it. An empty field falls back to the identifier, which the placeholder shows — so publishing an internal name is always a visible choice.</span></li>
|
||||
<li><span class="k">Address</span><span>The page address is fixed after creation and the record line carries the whole URL, click to copy. It is what gets pasted into a support article.</span></li>
|
||||
<li><span class="k">Copy</span><span>Buttons name the outcome: <b>Open incident</b>, <b>Post update</b>, <b>Schedule maintenance</b> — the same words the public timeline then shows.</span></li>
|
||||
</ul>
|
||||
</section>
|
||||
|
||||
</div>
|
||||
|
||||
<script>
|
||||
// 90 daily cells per component. Seeded rather than random so the mockup is
|
||||
// stable between reloads and reviewers are looking at the same picture.
|
||||
const PATTERNS = {
|
||||
api: { downs: [2, 22], maint: [], noData: 0 },
|
||||
hooks: { downs: [], maint: [], noData: 0 },
|
||||
dash: { downs: [38], maint: [], noData: 0 },
|
||||
db: { downs: [0, 1, 12, 13, 47], maint: [], noData: 0 },
|
||||
obj: { downs: [61], maint: [0], noData: 0 },
|
||||
new: { downs: [], maint: [], noData: 61 }
|
||||
};
|
||||
|
||||
document.querySelectorAll(".comp").forEach((comp) => {
|
||||
const p = PATTERNS[comp.dataset.bar];
|
||||
const bar = comp.querySelector(".bar");
|
||||
const frag = document.createDocumentFragment();
|
||||
for (let i = 89; i >= 0; i--) {
|
||||
const cell = document.createElement("span");
|
||||
let cls = "up";
|
||||
if (i >= 90 - p.noData) cls = "none";
|
||||
else if (p.maint.includes(i)) cls = "maint";
|
||||
else if (p.downs.includes(i)) cls = "down";
|
||||
cell.className = cls;
|
||||
cell.title = cls === "none" ? "no data" : cls === "maint" ? "maintenance" : cls === "down" ? "outage" : "operational";
|
||||
frag.appendChild(cell);
|
||||
}
|
||||
bar.appendChild(frag);
|
||||
});
|
||||
</script>
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,307 +0,0 @@
|
||||
# Multiple auth providers
|
||||
|
||||
Date: 2026-08-03
|
||||
|
||||
## Problem
|
||||
|
||||
An instance can configure exactly one OIDC provider. `instance_oidc` holds one
|
||||
document per instance, `/auth/oidc/start` takes no argument, and `/login`
|
||||
renders an unconditional "Sign in with your instance's SSO" button whether or
|
||||
not anything is configured behind it. Customers who federate with more than one
|
||||
identity source cannot, and customers who federate with none are shown a button
|
||||
that leads to an error.
|
||||
|
||||
## Goals
|
||||
|
||||
- N auth providers per instance, each independently enabled and named.
|
||||
- Login page renders one button per enabled provider, and none when there are
|
||||
none.
|
||||
- Local email/password login can be turned off per instance.
|
||||
- Presets for the common identity providers, so a customer supplies a tenant ID
|
||||
rather than an issuer URL.
|
||||
- Existing configured SSO keeps working across the upgrade with no customer
|
||||
action.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- SAML. Different protocol, metadata parsing and certificate handling; not in
|
||||
this work.
|
||||
- Per-provider role or group mapping. Provisioned users remain `member`, as
|
||||
today.
|
||||
- Provider-specific account linking. An email address is an email address; the
|
||||
existing instance-scoped lookup stands.
|
||||
|
||||
## Data model
|
||||
|
||||
New collection `auth_providers`, one document per provider:
|
||||
|
||||
```go
|
||||
type AuthProvider struct {
|
||||
ID bson.ObjectID `bson:"_id,omitempty" json:"_id,omitempty"`
|
||||
InstanceID string `bson:"instance_id" json:"instance_id"`
|
||||
ProviderID string `bson:"provider_id" json:"provider_id"`
|
||||
Name string `bson:"name" json:"name"`
|
||||
Kind string `bson:"kind" json:"kind"` // "oidc" | "oauth2"
|
||||
Preset string `bson:"preset" json:"preset"` // "" for custom
|
||||
Issuer string `bson:"issuer" json:"issuer"`
|
||||
ClientID string `bson:"client_id" json:"client_id"`
|
||||
ClientSecretEnc string `bson:"client_secret_enc,omitempty" json:"-"`
|
||||
Scopes []string `bson:"scopes" json:"scopes"`
|
||||
Enabled bool `bson:"enabled" json:"enabled"`
|
||||
CallbackNotice bool `bson:"callback_notice" json:"callback_notice"`
|
||||
Order int `bson:"order" json:"order"`
|
||||
CreatedAt time.Time `bson:"created_at" json:"created_at"`
|
||||
UpdatedAt time.Time `bson:"updated_at" json:"updated_at"`
|
||||
}
|
||||
```
|
||||
|
||||
`ProviderID` is a short random identifier, not the Mongo `_id`: it appears in
|
||||
the callback URL a customer pastes into their IdP, and an `_id` there would
|
||||
publish a database key.
|
||||
|
||||
Unique index on `(instance_id, provider_id)`. Index build is fatal on failure,
|
||||
matching `EnsureAuthIndexes` — a duplicate `provider_id` within an instance
|
||||
would make the callback ambiguous.
|
||||
|
||||
`ClientSecretEnc` is AES-256-GCM under `KEY_ENCRYPTION_KEY`, as
|
||||
`instance_oidc.client_secret_enc` is today, and is never serialised.
|
||||
|
||||
### Presets
|
||||
|
||||
A Go table in `server/internal/auth/presets.go`, not database rows — adding one
|
||||
is a commit, not a migration.
|
||||
|
||||
| Preset | Kind | Issuer | Input asked of the customer | Default scopes |
|
||||
| ------- | -------- | ------------------------------------------------- | --------------------------- | ----------------------------- |
|
||||
| `entra` | `oidc` | `https://login.microsoftonline.com/{tenant}/v2.0` | Directory (tenant) ID | `openid profile email` |
|
||||
| `google`| `oidc` | `https://accounts.google.com` | none | `openid profile email` |
|
||||
| `okta` | `oidc` | `https://{domain}/oauth2/default` | Okta org domain | `openid profile email` |
|
||||
| `github`| `oauth2` | n/a | none | `read:user user:email` |
|
||||
| `` (custom) | `oidc` | supplied verbatim | Issuer URL | `openid profile email` |
|
||||
|
||||
The issuer template is expanded server-side on save; the stored `Issuer` is
|
||||
always the resolved URL, so nothing downstream has to know a preset existed.
|
||||
|
||||
### Settings
|
||||
|
||||
`settings.local_login_enabled bool`, defaulting true. Absent on existing
|
||||
documents, and Go's zero value for `bool` is false, so the field is read through
|
||||
a `*bool` and a nil pointer means enabled. A plain `bool` would silently
|
||||
disable password login on every instance in the fleet at upgrade.
|
||||
|
||||
## Migration
|
||||
|
||||
`0005_auth_providers` — the next free number; `0004_instance_rename` is the
|
||||
highest recorded today. For each document in `instance_oidc`, insert one
|
||||
`auth_providers` document:
|
||||
|
||||
- `Name: "Single sign-on"`
|
||||
- `Preset: ""`, `Kind: "oidc"`
|
||||
- `Issuer`, `ClientID`, `Enabled` copied
|
||||
- `ClientSecretEnc` copied **verbatim**, not decrypted and re-encrypted — a
|
||||
migration that needs `KEY_ENCRYPTION_KEY` fails on an instance that has none
|
||||
and strands the SSO configuration.
|
||||
- `Scopes: ["openid", "profile", "email"]`, matching what `oidc.go` hardcodes
|
||||
today.
|
||||
- `ProviderID` freshly generated.
|
||||
- `CallbackNotice: true` — this provider's redirect URI has changed and an
|
||||
administrator has not yet acknowledged it. Set only by the migration; cleared
|
||||
by the settings UI. New providers are created `false`.
|
||||
|
||||
`instance_oidc` is left in place and no longer read. Idempotent by skipping any
|
||||
instance that already has an `auth_providers` document, so a re-run after a
|
||||
partial failure completes rather than duplicating.
|
||||
|
||||
## Auth flow
|
||||
|
||||
Routes:
|
||||
|
||||
```
|
||||
GET /auth/oidc/:providerId/start
|
||||
GET /auth/oidc/:providerId/callback
|
||||
```
|
||||
|
||||
The old unparameterised `/auth/oidc/start` and `/auth/oidc/callback` are
|
||||
**removed**, not retained. See Upgrade impact below — this breaks configured SSO
|
||||
until the customer updates their IdP, and that is accepted deliberately rather
|
||||
than carried as a compatibility path.
|
||||
|
||||
The state token in Redis stores `{instance_id, provider_id}` rather than the
|
||||
bare instance ID. The callback resolves its provider from the consumed state
|
||||
and cross-checks it against `:providerId` in the path, refusing a mismatch —
|
||||
the path alone is attacker-controlled, and the state is the half that was
|
||||
issued by the start handler.
|
||||
|
||||
`providerForInstance` becomes `providerFor(ctx, c, instanceID, providerID)`.
|
||||
The `go-oidc` provider cache keys on `provider_id`, not instance. Saving,
|
||||
disabling or deleting a provider evicts that key.
|
||||
|
||||
`redirectURL(c, providerID)` returns the one per-provider shape, and returns the
|
||||
same URL in the start and callback halves of a flow — an IdP rejects the token
|
||||
exchange if they differ.
|
||||
|
||||
### OIDC providers
|
||||
|
||||
Unchanged from the current implementation: `AuthCodeURL` with the stored
|
||||
scopes, exchange, `id_token` verified against the provider's key set with
|
||||
`ClientID` as audience, `email` and `name` claims extracted.
|
||||
|
||||
### GitHub (`kind: "oauth2"`)
|
||||
|
||||
GitHub is OAuth2 and issues no `id_token`, so it takes a separate branch:
|
||||
exchange the code, then `GET https://api.github.com/user/emails` with the access
|
||||
token and take the address that is both `primary` and `verified`. An
|
||||
unverified-only response is refused — an unverified address is not proof of
|
||||
control, and accepting one would let anyone holding a GitHub account claim any
|
||||
address in the instance. `name` comes from `GET https://api.github.com/user`.
|
||||
|
||||
Both branches converge on one function:
|
||||
|
||||
```go
|
||||
func completeSSOLogin(c *gin.Context, instanceID, email, name string) error
|
||||
```
|
||||
|
||||
which holds today's lookup-or-provision, session creation, `TouchLastLogin` and
|
||||
cookie set, verbatim. Email is lower-cased before lookup, and the lookup stays
|
||||
`GetUserInInstanceByEmail` — instance-scoped, as it is now.
|
||||
|
||||
### Licence gate
|
||||
|
||||
`services.GetLicenseState(instanceID).Feature("oidc")` continues to gate both
|
||||
the start and the callback, for every provider kind, and is checked on the
|
||||
callback against the instance named by the consumed state rather than the host.
|
||||
Unchanged behaviour, applied to more providers.
|
||||
|
||||
## REST API
|
||||
|
||||
Unauthenticated:
|
||||
|
||||
```
|
||||
GET /auth/providers
|
||||
-> {"local_enabled": true,
|
||||
"providers": [{"id": "...", "name": "...", "preset": "entra"}]}
|
||||
```
|
||||
|
||||
Instance is resolved from the host, as `/auth/bootstrap-status` already does.
|
||||
The response carries **no issuer, no client ID and no secret** — it is served to
|
||||
anyone who can reach the login page.
|
||||
|
||||
Session-authed, `owner|admin`, under `/api`:
|
||||
|
||||
```
|
||||
GET,POST /auth/providers
|
||||
PUT,DELETE /auth/providers/:id
|
||||
POST /auth/providers/:id/test
|
||||
```
|
||||
|
||||
`test` fetches the provider's discovery document (or, for GitHub, calls the API
|
||||
with the stored credentials) and reports reachability. It does not sign anyone
|
||||
in.
|
||||
|
||||
`GET,PUT /api/org/oidc` is removed along with the old auth routes. Its only
|
||||
caller is `OIDCCard.tsx`, which this work replaces, and a compatibility shim
|
||||
over a one-of-many model would have to invent which provider it means.
|
||||
|
||||
Every mutation writes an audit event, as every mutating path does.
|
||||
|
||||
### Lockout guards
|
||||
|
||||
Both refused with 409 and a distinct error code:
|
||||
|
||||
- `local_login_required` — disabling local login while zero providers are
|
||||
enabled.
|
||||
- `last_provider` — disabling or deleting the last enabled provider while local
|
||||
login is off.
|
||||
|
||||
These are enforced in the service layer, not the handler, so the two endpoints
|
||||
that can reach the condition cannot disagree.
|
||||
|
||||
## Frontend
|
||||
|
||||
### Settings
|
||||
|
||||
`web/components/settings/OIDCCard.tsx` becomes `AuthProvidersCard`, in the
|
||||
Access group of `/settings` where the OIDC card already lives. It renders the
|
||||
provider list with per-row enable toggle, edit, delete and drag ordering, an
|
||||
Add flow that asks for the preset first and then only the fields that preset
|
||||
needs, and the local-login toggle beneath the list. A guard violation surfaces
|
||||
the 409's message rather than a generic failure.
|
||||
|
||||
Every provider row shows its **callback URL** with click-to-copy — that is the
|
||||
value the customer pastes into their IdP, it now differs per provider, and after
|
||||
the upgrade every migrated provider needs it re-pasted. A migrated provider
|
||||
additionally carries a warning until an administrator dismisses it, naming the
|
||||
change and the URL. Dismissal is per provider, stored on the document.
|
||||
|
||||
### Login page
|
||||
|
||||
`web/app/login/page.tsx` calls `/auth/providers` on mount alongside the existing
|
||||
`bootstrapStatus` call, and renders on the result:
|
||||
|
||||
| `local_enabled` | providers | Rendered |
|
||||
| --------------- | --------- | --------------------------------------------------- |
|
||||
| true | none | Password form only. No divider, no buttons. |
|
||||
| true | some | Password form, divider, one button per provider. |
|
||||
| false | some | Buttons only. No form, no divider. |
|
||||
| false | none | Password form (see below). |
|
||||
|
||||
The last row cannot be reached through the API — the guards above prevent it —
|
||||
but a hand-edited database could produce it, and a login page that renders
|
||||
nothing at all is unrecoverable without database access. It therefore falls back
|
||||
to the password form.
|
||||
|
||||
The current unconditional SSO button and its "SSO must be enabled for this
|
||||
instance by an administrator" note are both removed; the button now only exists
|
||||
when it works.
|
||||
|
||||
Buttons are labelled with the provider's `Name` and carry the preset's icon
|
||||
where there is one, a neutral key glyph otherwise. Presets never override the
|
||||
name — a customer who calls their Entra provider "Staff" gets "Staff".
|
||||
|
||||
Errors keep the existing `/login?error=<code>` redirect convention.
|
||||
|
||||
## Testing
|
||||
|
||||
- Migration: an `instance_oidc` document produces one enabled provider with the
|
||||
ciphertext byte-identical; a re-run inserts nothing further.
|
||||
- `local_login_enabled` absent decodes as enabled.
|
||||
- Guards: both 409 paths, and the enable/disable sequences that approach them
|
||||
without crossing.
|
||||
- Per-provider callback: two providers in one instance, each resolving to its
|
||||
own configuration; a `provider_id` from another instance answers 404.
|
||||
- A callback whose `:providerId` disagrees with the consumed state is refused,
|
||||
and the state is consumed rather than left replayable.
|
||||
- The removed routes (`/auth/oidc/start`, `/auth/oidc/callback`,
|
||||
`/api/org/oidc`) answer 404.
|
||||
- GitHub: primary+verified selected; verified-only-absent refused.
|
||||
- `/auth/providers` response contains no issuer, client ID or secret.
|
||||
|
||||
## Upgrade impact
|
||||
|
||||
**This release breaks configured SSO until each customer updates their identity
|
||||
provider.** The old `/auth/oidc/callback` is gone, migrated providers are
|
||||
reachable only at `/auth/oidc/<providerId>/callback`, and an IdP still pointing
|
||||
at the old URL fails the flow.
|
||||
|
||||
It is a deliberate trade: one callback shape rather than two, no
|
||||
`legacy_callback` branch through `redirectURL`, and no permanently retained
|
||||
route whose only purpose is a single past upgrade.
|
||||
|
||||
Mitigations, in order of who sees them first:
|
||||
|
||||
- The settings card shows the new callback URL per provider with click-to-copy,
|
||||
and a migrated provider carries a dismissable warning naming the change.
|
||||
- The failure is visible rather than silent: an IdP rejects the redirect URI
|
||||
before Vantage is reached, so the customer sees their own provider's error.
|
||||
- Local password login is unaffected, so no instance is locked out — an
|
||||
administrator can always sign in to fix the URL. This is why
|
||||
`local_login_enabled` defaults to true and why nothing in this migration
|
||||
turns it off.
|
||||
- Release notes and `docsite/docs/vantage/settings.md` state the required
|
||||
action.
|
||||
|
||||
## Deployment notes
|
||||
|
||||
No new environment variables. No agent change. `KEY_ENCRYPTION_KEY` is already
|
||||
required wherever OIDC was configured, and the migration does not add a
|
||||
dependency on it.
|
||||
@@ -1,235 +0,0 @@
|
||||
# Server tags and scheduled workflows
|
||||
|
||||
Date: 2026-08-04
|
||||
|
||||
Two features, designed together because the second is worth much less without
|
||||
the first. Tags make a target set describable; schedules make it recur. A
|
||||
nightly job that patches "everything tagged `env:staging`" needs both halves,
|
||||
and neither half is large on its own.
|
||||
|
||||
---
|
||||
|
||||
## Part A — Server tags
|
||||
|
||||
### Model
|
||||
|
||||
`models.Server` gains one field:
|
||||
|
||||
```go
|
||||
Tags map[string]string `bson:"tags,omitempty" json:"tags,omitempty"`
|
||||
```
|
||||
|
||||
Keys and values are lowercase `[a-z0-9_-]`. Keys are capped at 32 characters,
|
||||
values at 64, and a server holds at most 20 tags. Validation lives in the
|
||||
service layer rather than the handler, so the tag endpoint, the server-create
|
||||
path and anything added later cannot disagree about what a valid tag is.
|
||||
|
||||
There is **no `tags` collection.** A tag is a property of a server, not an
|
||||
entity with a lifecycle: a registry would need reference counting to know when
|
||||
a tag stopped existing, and garbage collection to act on it, which is work
|
||||
bought for nothing. The list of known keys and values that the UI offers for
|
||||
autocomplete is a distinct aggregation over `servers`, cached for 60 seconds —
|
||||
the same treatment org lookups already get.
|
||||
|
||||
No reserved keys ship in this change. If inventory-derived tags (`os`, `arch`)
|
||||
are added later they take a `sys:` key prefix, so a user tag written today can
|
||||
never collide with a system tag invented tomorrow.
|
||||
|
||||
Index: `{instance_id: 1, "tags.$**": 1}` — a wildcard index over the tag
|
||||
subdocument, because the queried key is chosen by the user at request time and
|
||||
cannot be named in advance.
|
||||
|
||||
### API
|
||||
|
||||
```
|
||||
PUT /api/servers/:id/tags # replace the whole map
|
||||
GET /api/servers/tags # known keys and values, for pickers
|
||||
GET /api/servers?tag=env:prod # repeatable; AND across keys
|
||||
```
|
||||
|
||||
`PUT` replaces the entire map rather than patching one tag. A tag set is small
|
||||
enough that sending all of it is free, and last-write-wins over a whole map is
|
||||
easier to reason about than merge semantics between two people editing the same
|
||||
server. The audit event records the map before and after.
|
||||
|
||||
`?tag=` is repeatable and ANDs: `?tag=env:prod&tag=role:web` matches servers
|
||||
carrying both. A malformed value (no colon, unknown characters) is a 400 rather
|
||||
than a silent empty result — a filter that matches nothing and a filter that is
|
||||
nonsense look identical in a list, and only one of them is the user's fault.
|
||||
|
||||
### Targeting
|
||||
|
||||
`models.Workflow` gains `TargetTags map[string]string` beside the existing
|
||||
`TargetServerIDs`. One function in `services` resolves them:
|
||||
|
||||
```go
|
||||
ResolveTargets(ctx, instanceID string, ids []string, tags map[string]string) ([]Server, error)
|
||||
```
|
||||
|
||||
- Result is the **distinct union** of the explicit IDs and the tag matches.
|
||||
- Tag matching ANDs across keys.
|
||||
- Offline servers are included. The dispatcher already answers 503 per server,
|
||||
and a patch run that silently omits an unreachable machine is worse than one
|
||||
that visibly fails on it.
|
||||
- Empty IDs **and** empty tags returns `ErrNoTargets` (400). A workflow that
|
||||
matches nothing must say so rather than report success over zero servers.
|
||||
|
||||
The resolved set is snapshotted into `WorkflowRun.ServerRuns` exactly as today.
|
||||
History records what actually ran, not what the selector would match when the
|
||||
run is later read back — the same reason `steps_snapshot` exists.
|
||||
|
||||
### Frontend
|
||||
|
||||
- **Server detail**: tag chips in the header with an inline editor. Keys
|
||||
autocomplete from `GET /api/servers/tags`, values autocomplete per key.
|
||||
- **`/servers`**: a filter bar that reads and writes the same `?tag=` query
|
||||
params the API takes, so a filtered fleet view is a URL someone can send.
|
||||
- **Workflow designer**: a target section holding both inputs, with a live
|
||||
"runs on 14 servers" readout that lists them on hover. The union model costs
|
||||
us the at-a-glance answer to "what will this touch"; this readout buys it
|
||||
back, and it is the reason the union is acceptable.
|
||||
|
||||
---
|
||||
|
||||
## Part B — Scheduled workflows
|
||||
|
||||
### Model
|
||||
|
||||
```go
|
||||
type Schedule struct {
|
||||
Enabled bool `bson:"enabled" json:"enabled"`
|
||||
Cron string `bson:"cron" json:"cron"` // 5-field
|
||||
TZ string `bson:"tz" json:"tz"` // IANA name
|
||||
}
|
||||
|
||||
type Skip struct {
|
||||
Reason string `bson:"reason" json:"reason"` // "missed" | "already_running"
|
||||
Due time.Time `bson:"due" json:"due"`
|
||||
At time.Time `bson:"at" json:"at"`
|
||||
}
|
||||
```
|
||||
|
||||
On `Workflow`:
|
||||
|
||||
```go
|
||||
Schedule *Schedule `bson:"schedule,omitempty"`
|
||||
NextRunAt *time.Time `bson:"next_run_at,omitempty"` // UTC, indexed
|
||||
LastRunAt *time.Time `bson:"last_run_at,omitempty"`
|
||||
LastSkipped *Skip `bson:"last_skipped,omitempty"`
|
||||
```
|
||||
|
||||
`next_run_at` is **persisted, not held in memory.** A leader handover between
|
||||
computing the next occurrence and firing it would otherwise either lose the
|
||||
occurrence or fire it twice. Coordination state has to live where every replica
|
||||
can see it — the same argument that put `workflow_log_seq` in MongoDB.
|
||||
|
||||
Cron parsing uses `robfig/cron/v3`'s **parser only** — `Parse` and
|
||||
`Next(time)`. Its scheduler and goroutines are not used; the loop below is ours
|
||||
and has to be, because it runs under the leader lock.
|
||||
|
||||
**Alpine ships no tzdata.** `server/Dockerfile` builds a slim image, so
|
||||
`time.LoadLocation("Europe/London")` returns an error and every schedule
|
||||
falls back to UTC — an hour wrong for half the year, in the direction nobody
|
||||
notices until a maintenance window lands in business hours. `main` therefore
|
||||
imports `_ "time/tzdata"`, embedding the database in the binary. Zone names are
|
||||
also validated at save time, so an unknown zone is a 400 rather than a surprise
|
||||
at 2am.
|
||||
|
||||
### Scheduler
|
||||
|
||||
A new `server/internal/workflowsched` package, started inside the **existing**
|
||||
`bus.RunAsLeader("housekeeping", …)` alongside `monitorsched`, `StartReaper`
|
||||
and the sweepers. One role, one lock. It takes the same cancellable context and
|
||||
returns the instant leadership is lost.
|
||||
|
||||
The loop ticks every 30 seconds:
|
||||
|
||||
1. `find({schedule.enabled: true, next_run_at: {$lte: now}})`.
|
||||
2. **Claim atomically.** `findOneAndUpdate` matching the document *and* its
|
||||
current `next_run_at`, setting the recomputed next occurrence. A process
|
||||
that reaches the same document after another has claimed it matches nothing
|
||||
and does nothing. The claim is what makes this correct; the leader lock only
|
||||
makes it cheap.
|
||||
3. **Grace check.** If `now - due > 1h`, record
|
||||
`last_skipped{reason: "missed"}`, write an audit event, and do not run. A
|
||||
job missed by ten minutes during a deploy should still run; one missed by
|
||||
two days should not fire at lunchtime.
|
||||
4. **Overlap check.** If a run for this workflow is still active, record
|
||||
`last_skipped{reason: "already_running"}`, audit, and do not run. A patch
|
||||
workflow must never run twice at once, and a silent skip is how a week goes
|
||||
by before anyone notices nothing ran.
|
||||
5. Otherwise start the run through the **same** `RunWorkflow` path a person
|
||||
uses, with `TriggeredBy: "schedule"`.
|
||||
|
||||
Step 5 is the design. A scheduled run is an ordinary run with a different
|
||||
trigger: no second dispatch path, no second snapshot format, and the run detail
|
||||
page needs no changes to display one.
|
||||
|
||||
### API
|
||||
|
||||
```
|
||||
PUT /api/workflows/:id/schedule # {enabled, cron, tz}
|
||||
GET /api/workflows/:id/schedule/preview?cron=…&tz=… # next 3 occurrences
|
||||
```
|
||||
|
||||
`PUT` validates the expression and the zone, then computes and stores
|
||||
`next_run_at`. The preview endpoint exists so the browser and the scheduler
|
||||
agree on what a cron string means — a client-side cron parser that disagrees
|
||||
with the server by one field is a bug found in production, at night.
|
||||
|
||||
### Frontend
|
||||
|
||||
- **Workflow page**: a schedule card with preset buttons (hourly, nightly at
|
||||
HH:MM, weekly on DAY at HH:MM) that write cron underneath, a raw cron field
|
||||
for anything else, a timezone select, and the next three occurrences rendered
|
||||
from the preview endpoint in mono.
|
||||
- **Workflows list**: a schedule chip and the next run as relative time.
|
||||
- **Skips are surfaced**, not just stored: a warning line reading
|
||||
"Skipped Sun 02:00 — previous run still active". Recording a reason nobody
|
||||
reads is the same as not recording one.
|
||||
|
||||
---
|
||||
|
||||
## Out of scope
|
||||
|
||||
**Notification on scheduled-run failure.** It needs the monitor channel
|
||||
machinery pointed at workflow outcomes and its own answer to what counts as
|
||||
failure — a non-zero exit on a step with `on_failure: continue` is not
|
||||
obviously an alert. Visibility in this change is the run list and the recorded
|
||||
skip reason. Excluded deliberately, not overlooked.
|
||||
|
||||
**Tag-scoped permissions.** Roles stay instance-wide. Tags describe servers;
|
||||
they do not yet gate who may act on them.
|
||||
|
||||
**Inventory-derived tags.** Reserved via the `sys:` prefix, not implemented.
|
||||
|
||||
---
|
||||
|
||||
## Migration and compatibility
|
||||
|
||||
No migration is required. `Tags`, `TargetTags` and `Schedule` are all
|
||||
`omitempty` and absent means what it meant before: no tags, no selector, no
|
||||
schedule. Existing workflows keep their explicit server lists and behave
|
||||
identically.
|
||||
|
||||
The wildcard tag index and the `next_run_at` index are declared by a new
|
||||
`EnsureServerIndexes`, following the convention `EnsureSecretIndexes` and
|
||||
`EnsureWorkflowIndexes` already set: it warns rather than aborting boot,
|
||||
because a missing index degrades
|
||||
tag filtering to a collection scan on a small collection rather than breaking
|
||||
the fleet list.
|
||||
|
||||
## Testing
|
||||
|
||||
- `ResolveTargets`: union deduplicates; AND across tag keys; empty/empty
|
||||
returns `ErrNoTargets`; offline servers are included.
|
||||
- Tag validation: charset, length caps, tag count cap, malformed `?tag=` is a
|
||||
400.
|
||||
- Schedule validation: bad cron and unknown zone both 400; `next_run_at` is
|
||||
computed in the stored zone, verified across a DST boundary.
|
||||
- Scheduler claim: two concurrent claims of the same due workflow start exactly
|
||||
one run.
|
||||
- Grace window: due 10 minutes ago runs; due 2 hours ago records `missed`.
|
||||
- Overlap: an active run yields `already_running` and no second run.
|
||||
- Preview endpoint and the scheduler agree on the next occurrence for a table
|
||||
of expressions, including a DST-crossing one.
|
||||
@@ -1,538 +0,0 @@
|
||||
# Package inventory and CVE findings
|
||||
|
||||
Date: 2026-08-06
|
||||
|
||||
Agents report the packages installed on each server. The control plane matches
|
||||
them against distro security feeds and raises findings that link straight to
|
||||
the patching path that already exists. A finding nobody can fix today can be
|
||||
accepted with a reason and an expiry date rather than sitting red forever.
|
||||
|
||||
This is one of four sub-projects sketched together and deliberately separated:
|
||||
|
||||
| # | Sub-project | Depends on |
|
||||
| - | ----------- | ---------- |
|
||||
| A | **Package inventory + CVE findings** — this spec | nothing |
|
||||
| B | Container/service registry | nothing |
|
||||
| C | Container image scanning | A and B |
|
||||
| D | Compliance profiles (baseline assertions) | shares A's findings UI only |
|
||||
|
||||
A and B are independent of one another. C is the joiner and must not be
|
||||
designed before both exist. D shares a page with A and nothing else — a
|
||||
different collector, a different evaluation model and a different remediation
|
||||
story — so folding it in here would double the size for no shared machinery.
|
||||
|
||||
Scope of this spec is **A, Linux only.** Windows needs a separate source
|
||||
(MSRC CVRF), a separate collector (`Get-HotFix` plus registry) and a KB
|
||||
supersedence matcher that shares no code with the Linux path. That matches the
|
||||
existing position that Windows agents are second-class by design, and the six
|
||||
package managers `updates.go` already detects cover the whole Linux surface.
|
||||
|
||||
---
|
||||
|
||||
## The trap this design is built around
|
||||
|
||||
Distributions **backport** security fixes without changing the upstream
|
||||
version. Ubuntu ships `openssl 3.0.2-0ubuntu1.15` patched against
|
||||
CVE-2023-0286; NVD says version 3.0.2 is vulnerable. Matching installed
|
||||
versions against NVD or CPE ranges therefore reports a fleet full of criticals
|
||||
that are all already fixed.
|
||||
|
||||
That is not merely noisy. It is fatal to the feature: once the first report is
|
||||
mostly wrong, nobody reads the second one, and a genuine finding is lost in the
|
||||
noise it created. Everything below follows from refusing to make that mistake.
|
||||
|
||||
The correct source is the **distribution's own security feed**, keyed on the
|
||||
distribution's own version string — Debian and Ubuntu OVAL/USN, Red Hat OVAL
|
||||
v2, Alpine secdb. `trivy-db` is those feeds pre-merged into one BoltDB
|
||||
artifact, rebuilt every six hours and published as an OCI artifact.
|
||||
|
||||
---
|
||||
|
||||
## Where the vulnerability data comes from
|
||||
|
||||
`trivy-db`, pulled server-side from `ghcr.io/aquasecurity/trivy-db:2`.
|
||||
|
||||
The alternative considered was querying OSV.dev per scan, which needs no
|
||||
storage and no puller. It was rejected on two counts: it requires outbound
|
||||
internet on every scan, which breaks air-gapped installs; and it sends the
|
||||
package list of a customer's entire fleet to a third party. The audience most
|
||||
likely to buy vulnerability scanning is the audience least willing to do that.
|
||||
|
||||
The blob is roughly 50MB, read-only, reproducible, and identified by a version
|
||||
number. **It is not stored in Mongo and not written to `/data`** —
|
||||
`server.persistence` defaults to off and nothing writes to `/data` any more.
|
||||
It does not need durable storage: whichever pod needs it pulls it to its own
|
||||
ephemeral temp directory. Nothing shared, nothing to back up, nothing to
|
||||
migrate.
|
||||
|
||||
`VANTAGE_TRIVY_DB_REF` overrides the default reference so a customer can mirror
|
||||
the artifact into their own registry. It also covers the anonymous ghcr rate
|
||||
limit, which the six-hourly pull cadence already makes unlikely to bite.
|
||||
|
||||
---
|
||||
|
||||
## Only the leader matches
|
||||
|
||||
This is the crux, and it falls out of the replica model already in the
|
||||
codebase.
|
||||
|
||||
Two things trigger matching, and they happen on different pods:
|
||||
|
||||
1. a fleet-wide rescan when `trivy-db` updates — naturally the leader's job
|
||||
2. a server's package list changing — handled by whichever pod holds *that
|
||||
agent's* command stream
|
||||
|
||||
If (2) matched inline, **every replica would need the 50MB database resident**,
|
||||
and a database refresh would have N pods racing to rescan the same fleet and N
|
||||
digests reaching the customer. That is the exact failure `RunAsLeader` exists
|
||||
to prevent, and it is the same argument that put `monitorsched` behind the
|
||||
lock.
|
||||
|
||||
So `ReportPackages` does not match. It upserts the package list and sets
|
||||
`scan_pending: true`. That is all it does.
|
||||
|
||||
`server/internal/vulnsched` then runs inside the **existing**
|
||||
`bus.RunAsLeader("housekeeping", …)` alongside `monitorsched`,
|
||||
`workflowsched` and the sweepers — one role, one lock. Every 60 seconds it:
|
||||
|
||||
1. pulls `trivy-db` if the local copy is older than six hours
|
||||
2. if the pulled version differs from `vulndb_meta.db_version`, marks **every**
|
||||
server `scan_pending`
|
||||
3. matches all `scan_pending` servers, clears the flag, diffs against existing
|
||||
findings
|
||||
4. emits **one** digest per tick covering everything newly opened
|
||||
|
||||
Step 4 is why batching is structural rather than bolted on. A `trivy-db`
|
||||
refresh can open several hundred findings across a fleet at once; one message
|
||||
per finding would rate-limit the webhook or get the channel muted, and either
|
||||
way the customer stops receiving the alerts they are paying for. The tick is
|
||||
already the natural batch boundary, so **the failure cannot occur by
|
||||
construction** rather than by a debounce someone has to maintain.
|
||||
|
||||
`scan_pending` lives on the document rather than in memory, for the same reason
|
||||
`next_run_at` and `workflow_log_seq` do: a leader handover between marking and
|
||||
scanning would otherwise lose it. A handover costs the new leader one re-pull
|
||||
of the database.
|
||||
|
||||
The cost of this indirection is up to 60 seconds between an agent reporting a
|
||||
changed package set and its findings updating. For vulnerability data that is
|
||||
nothing, and it buys a single matching path instead of two.
|
||||
|
||||
---
|
||||
|
||||
## Components
|
||||
|
||||
```
|
||||
agent/internal/packages/ collect installed packages + /etc/os-release
|
||||
proto/ ReportPackages RPC
|
||||
server/internal/vulndb/ puller, BoltDB access, matcher
|
||||
server/internal/vulnsched/ leader-owned tick: pull, scan, digest
|
||||
server/internal/services/ findings, acceptance, alert rules
|
||||
web/app/(app)/vulnerabilities/ fleet board; plus two server-detail tabs
|
||||
```
|
||||
|
||||
`vulnsched` takes the dependencies it needs — `LogEvent` and the notification
|
||||
dispatch — as a `vulnsched.Deps` injected from `main.go`, following
|
||||
`workflowsched`. The manual rescan endpoint does not call into `vulnsched` at
|
||||
all: it sets `scan_pending` on every server and lets the next tick find them,
|
||||
so there is no path by which `services` imports the scheduler and no cycle to
|
||||
avoid later.
|
||||
|
||||
---
|
||||
|
||||
## The wire path
|
||||
|
||||
A new `ReportPackages` RPC on the agent's existing hourly loop — the same
|
||||
`runUpdateCheck` cadence, reusing `updates.go`'s `detectPM()`.
|
||||
|
||||
```protobuf
|
||||
rpc ReportPackages(ReportPackagesRequest) returns (ReportPackagesResponse);
|
||||
|
||||
message ReportPackagesRequest {
|
||||
string server_id = 1;
|
||||
string agent_token = 2;
|
||||
string hash = 3; // sha256 of the sorted list
|
||||
OSRelease os = 4;
|
||||
repeated InstalledPackage packages = 5; // omitted when only offering a hash
|
||||
}
|
||||
|
||||
message ReportPackagesResponse {
|
||||
bool need_full = 1; // hash differs; resend with packages populated
|
||||
}
|
||||
```
|
||||
|
||||
The agent calls once with `packages` empty. `need_full` true means the hash
|
||||
differs from what the server holds, and the agent immediately calls again with
|
||||
the list populated.
|
||||
|
||||
The agent sends a SHA-256 of its sorted package list first. If it matches what
|
||||
the server already holds, the server answers `unchanged` and the ~150KB body is
|
||||
never sent. A machine's package set changes rarely, so almost every hour costs
|
||||
one small message, and the rare changed hour costs one extra round trip.
|
||||
|
||||
Folding the list into the existing 15-minute `InventoryReport` static snapshot
|
||||
was rejected: it would re-send ~150KB per server every 15 minutes regardless of
|
||||
change, roughly 40MB/hour of gRPC traffic on a 100-server fleet to transmit
|
||||
data that is almost always identical.
|
||||
|
||||
---
|
||||
|
||||
## Data model
|
||||
|
||||
Four new collections. Every one carries `instance_id` except `vulndb_meta`,
|
||||
which is explained below.
|
||||
|
||||
### `server_packages` — one document per server, not per package
|
||||
|
||||
```go
|
||||
type ServerPackages struct {
|
||||
ID primitive.ObjectID `bson:"_id"`
|
||||
InstanceID primitive.ObjectID `bson:"instance_id"`
|
||||
ServerID string `bson:"server_id"`
|
||||
OS OSRelease `bson:"os"` // family, version_id, arch
|
||||
Hash string `bson:"hash"` // sha256 of the sorted list
|
||||
Packages []InstalledPackage `bson:"packages"`
|
||||
CollectedAt time.Time `bson:"collected_at"`
|
||||
ScanPending bool `bson:"scan_pending"`
|
||||
ScannedAt time.Time `bson:"scanned_at"`
|
||||
Status string `bson:"status"` // ok | unsupported
|
||||
DBVersion int `bson:"db_version"` // last matched against
|
||||
}
|
||||
|
||||
type InstalledPackage struct {
|
||||
Name string `bson:"name"`
|
||||
Version string `bson:"version"` // distro version string, verbatim
|
||||
Epoch int `bson:"epoch,omitempty"`
|
||||
Arch string `bson:"arch"`
|
||||
SourceName string `bson:"source_name,omitempty"`
|
||||
}
|
||||
```
|
||||
|
||||
One document rather than two thousand is what makes a report a **single atomic
|
||||
upsert with no delta logic** — the hash already established that something
|
||||
changed, so there is nothing to reconcile field by field. A typical Linux host
|
||||
lands near 150KB, comfortably inside the 16MB document limit.
|
||||
|
||||
Indexes: `{instance_id, server_id}` unique, and a multikey
|
||||
`{instance_id, "packages.name"}` for fleet-wide package search.
|
||||
|
||||
`SourceName` is not decoration. **Debian and Ubuntu advisories are keyed on the
|
||||
source package**: a CVE against `openssl` covers the binaries `libssl3`,
|
||||
`openssl` and `libssl-dev`, so matching on binary name alone misses two of the
|
||||
three.
|
||||
|
||||
`OS.VersionID` selects the feed. Ubuntu 22.04 and 24.04 publish different fixed
|
||||
versions for the same CVE, so a scan without it is guesswork.
|
||||
|
||||
### `vuln_findings` — one document per (server, CVE, package)
|
||||
|
||||
```go
|
||||
type VulnFinding struct {
|
||||
ID primitive.ObjectID `bson:"_id"`
|
||||
InstanceID primitive.ObjectID `bson:"instance_id"`
|
||||
ServerID string `bson:"server_id"`
|
||||
|
||||
CVEID string `bson:"cve_id"`
|
||||
PackageName string `bson:"package_name"`
|
||||
Installed string `bson:"installed_version"`
|
||||
FixedIn string `bson:"fixed_in,omitempty"`
|
||||
Severity string `bson:"severity"`
|
||||
CVSSScore float64 `bson:"cvss_score,omitempty"`
|
||||
Title string `bson:"title,omitempty"`
|
||||
References []string `bson:"references,omitempty"`
|
||||
|
||||
State string `bson:"state"` // open | fixed | accepted
|
||||
FirstSeen time.Time `bson:"first_seen"`
|
||||
LastSeen time.Time `bson:"last_seen"`
|
||||
FixedAt *time.Time `bson:"fixed_at,omitempty"`
|
||||
Accepted *Acceptance `bson:"accepted,omitempty"`
|
||||
}
|
||||
|
||||
type Acceptance struct {
|
||||
By primitive.ObjectID `bson:"by"`
|
||||
Reason string `bson:"reason"`
|
||||
Until time.Time `bson:"until"`
|
||||
At time.Time `bson:"at"`
|
||||
}
|
||||
```
|
||||
|
||||
Unique on `{instance_id, server_id, cve_id, package_name}`. That key is what
|
||||
makes a rescan an idempotent upsert rather than a duplicate factory, and it is
|
||||
what lets `first_seen` survive across scans. Query index
|
||||
`{instance_id, state, severity}`.
|
||||
|
||||
**An empty `FixedIn` is a real and common state** and must never be conflated
|
||||
with "not vulnerable". A CVE with no vendor fix published yet is exactly the
|
||||
finding people most need to see, and also the one that most needs acceptance,
|
||||
because there is nothing to patch.
|
||||
|
||||
Findings are **not deleted when a package is patched**. State moves to `fixed`
|
||||
with `fixed_at` set, so "what did we remediate last quarter" remains
|
||||
answerable — which is the question an auditor asks.
|
||||
|
||||
### `vulndb_meta` — singleton, deliberately unscoped
|
||||
|
||||
`db_version`, `pulled_at`, `last_full_scan_at`, `last_error`. It carries no
|
||||
`instance_id` because the vulnerability database is a property of the
|
||||
deployment, not of a tenant. Same reasoning as `migrations`.
|
||||
|
||||
### `vuln_alert_rules`
|
||||
|
||||
`instance_id`, `name`, `enabled`, `min_severity`, `tags map[string]string`,
|
||||
`channel_ids []`, timestamps.
|
||||
|
||||
The tag filter resolves through **`services.ResolveTargets`**, not a second
|
||||
matcher. That function is already the single answer to which servers a
|
||||
selector touches, and an alert rule that disagreed with a workflow about what
|
||||
`env:prod` means would be worse than having no filter at all.
|
||||
|
||||
---
|
||||
|
||||
## The matching engine
|
||||
|
||||
```
|
||||
server/internal/vulndb/
|
||||
pull.go OCI fetch → temp dir, version compare against vulndb_meta
|
||||
db.go BoltDB open, advisory lookup by (ecosystem, source, version)
|
||||
match.go per-family matching, severity resolution
|
||||
version.go dispatch to deb/rpm/apk comparator by OS family
|
||||
```
|
||||
|
||||
Dependencies: `github.com/aquasecurity/trivy-db` for the BoltDB schema, plus
|
||||
`go-deb-version`, `go-rpm-version` and `go-apk-version` — each a small
|
||||
standalone module doing one job. The roughly 200 lines of per-distro advisory
|
||||
lookup are ours.
|
||||
|
||||
Importing `trivy` itself was rejected: it would pull a very large transitive
|
||||
dependency tree into the server binary for one feature, and its Go API carries
|
||||
no stability guarantee across minor versions. Shelling out to the `trivy`
|
||||
binary against a generated SBOM was rejected for shipping a second binary in
|
||||
the image and turning a library call into subprocess lifecycle, timeouts and
|
||||
output-format drift.
|
||||
|
||||
### Why the comparators are bought rather than written
|
||||
|
||||
Version ordering is where this feature lives or dies, and its failure mode is
|
||||
silent. `dpkg` ordering has epochs, and `~` sorts *before* the empty string, so
|
||||
`3.0.2-0ubuntu1.15~rc1` precedes `3.0.2-0ubuntu1.15`. `rpmvercmp` has its own
|
||||
segment rules and treats `~` and `^` differently again. A `strings.Compare` or
|
||||
a semver parse orders `1.9` above `1.10` and reports a vulnerable fleet as
|
||||
clean — a false negative, which nobody notices until it matters.
|
||||
|
||||
### Scanning one server
|
||||
|
||||
1. Load `server_packages`; resolve OS family and version to a `trivy-db`
|
||||
ecosystem.
|
||||
2. **Unsupported ecosystem → record `status: unsupported`, clear the flag,
|
||||
write no findings.**
|
||||
3. For each package: resolve source name, look up advisories, compare versions.
|
||||
4. Upsert vulnerable results as `open`, preserving `first_seen`.
|
||||
5. Any currently-`open` finding absent from this result set → `fixed`, stamp
|
||||
`fixed_at`.
|
||||
6. Any `accepted` finding past its `until` → back to `open`.
|
||||
7. Clear `scan_pending`, stamp `scanned_at` and `db_version`.
|
||||
|
||||
Steps 5 and 6 must run in that order, so a finding that is both absent and
|
||||
expired settles as `fixed` rather than reopening on a package that no longer
|
||||
carries it.
|
||||
|
||||
Step 2 matters as much as any of the matching. Arch has no feed in `trivy-db`,
|
||||
so an Arch host must report **unsupported**, never "0 findings". Reporting
|
||||
clean when the truth is unknown is the same class of lie as a silently stale
|
||||
database, and it is the reason `vulndb_meta.pulled_at` appears on screen rather
|
||||
than only in a log.
|
||||
|
||||
### Severity
|
||||
|
||||
Resolved **vendor → NVD → unknown**, in that order, never invented.
|
||||
|
||||
This will surface as "why is this critical CVE marked low", and the answer is
|
||||
that Debian and Red Hat routinely downgrade an NVD score because the vulnerable
|
||||
code path is not reachable in their build. Their rating is the accurate one for
|
||||
that package, and showing NVD's above it would manufacture work that does not
|
||||
need doing.
|
||||
|
||||
---
|
||||
|
||||
## Findings lifecycle
|
||||
|
||||
`open | fixed | accepted`.
|
||||
|
||||
An accepted finding is suppressed from counts and alerts until its `until`
|
||||
date, then reopens automatically. A reason is required.
|
||||
|
||||
Acceptance with a mandatory expiry, rather than permanent dismissal, is what
|
||||
keeps the feature usable in both directions. Without any acceptance mechanism,
|
||||
a kernel CVE awaiting a reboot window sits red indefinitely and trains people
|
||||
to ignore the page. With permanent dismissal, accepted findings accumulate
|
||||
silently and nobody revisits them — the dismissal list becomes where risk goes
|
||||
to be forgotten, which is precisely what an auditor asks to see.
|
||||
|
||||
Retention: `settings.vuln_finding_retention_days`, a `*int` on the same pattern
|
||||
as `workflow_log_retention_days` — nil means 90 days, 0 means forever. Only
|
||||
`fixed` findings are swept, by a `StartVulnSweeper` inside the same
|
||||
`RunAsLeader("housekeeping", …)` as the existing sweepers. `open` and
|
||||
`accepted` findings are never swept at any setting.
|
||||
|
||||
---
|
||||
|
||||
## Alerting
|
||||
|
||||
Per-org rules over the existing `notification_channels`: severity threshold,
|
||||
optional tag filter, target channels.
|
||||
|
||||
A rescan emits one message summarising what newly opened — "12 new critical
|
||||
across 4 servers" — never one message per finding. See the leader section for
|
||||
why the tick boundary makes this structural.
|
||||
|
||||
Modelling findings as a monitor type was rejected. It would reuse monitors'
|
||||
state machine and channel wiring for free, but monitors are up/down for one
|
||||
endpoint with retries and hourly rollups, none of which means anything for a
|
||||
CVE; most fields would be disabled in the UI and the uptime graphs would be
|
||||
polluted with a signal that is not uptime.
|
||||
|
||||
This adds one `notify` payload type and a `vuln_digest.html.tmpl` /
|
||||
`vuln_digest.txt.tmpl` pair in `shared/mail`. Note that `shared/mail` templates
|
||||
are parsed in `init()`, so a mistyped field is a boot-time panic — CLAUDE.md
|
||||
describes a `render_test.go` guarding against exactly this, but **that file does
|
||||
not exist**; the repository has no Go tests at all, and by instruction this
|
||||
feature adds none. The template pair must therefore be verified by starting the
|
||||
binary and sending one digest through a real channel.
|
||||
|
||||
---
|
||||
|
||||
## Entitlement
|
||||
|
||||
The feature name is `vuln_scanning`, and it crosses the two services the way
|
||||
every other feature does:
|
||||
|
||||
- **admin** carries it as a per-instance entitlement toggle, so it can later be
|
||||
priced as a catalogue `feature` component without a second migration;
|
||||
- **the licence** snapshots it into `License.Features []string` at issue time;
|
||||
- **the server** asks `lic.HasFeature("vuln_scanning")` and never switches on
|
||||
tier, so changing what a tier includes needs no server release.
|
||||
|
||||
Off on Free.
|
||||
|
||||
**The gate is checked at `ReportPackages`, not at display.** Gating only the UI
|
||||
would still pay every write cost, and storage is the expensive half.
|
||||
|
||||
The agent learns of it through the existing 30-second `SyncKeys` poll:
|
||||
`SyncResponse` gains a `collect_packages` bool, and the hourly loop skips
|
||||
collection entirely when it is false. So an ungated instance produces no
|
||||
collection, no gRPC body, no document and no storage. `ReportPackages` still
|
||||
re-checks the entitlement server-side and refuses — the agent flag is an
|
||||
optimisation, the server check is the boundary.
|
||||
|
||||
Turning the feature off does not delete existing findings; they stop being
|
||||
served and stop updating. Deletion is the instance-deletion path's job.
|
||||
|
||||
---
|
||||
|
||||
## REST API
|
||||
|
||||
```
|
||||
GET /api/vulnerabilities # filter: severity, state, server, tags
|
||||
GET /api/vulnerabilities/summary # severity counts + database freshness
|
||||
POST /api/vulnerabilities/rescan # marks all scan_pending (owner|admin)
|
||||
POST /api/vulnerabilities/:id/accept # reason + until (owner|admin)
|
||||
DELETE /api/vulnerabilities/:id/accept # (owner|admin)
|
||||
GET /api/servers/:id/vulnerabilities
|
||||
GET /api/servers/:id/packages
|
||||
GET /api/packages/search?name= # fleet-wide
|
||||
GET,POST /api/vuln-rules · PUT,DELETE /api/vuln-rules/:id
|
||||
```
|
||||
|
||||
Every mutating path writes an audit event, as all of them do. Acceptance is the
|
||||
one decision people will be asked to justify, so `by`, `reason`, `until` and
|
||||
`at` land in the audit record and not only on the document.
|
||||
|
||||
---
|
||||
|
||||
## UI
|
||||
|
||||
`/vulnerabilities` is a fleet board **grouped by CVE** — one row per CVE with
|
||||
an affected-server count, expandable to the individual servers. The same CVE
|
||||
across 40 servers is one decision, and a flat list of findings makes it look
|
||||
like forty.
|
||||
|
||||
Server detail gains **Vulnerabilities** and **Packages** tabs. Alert rules go
|
||||
on `/settings/notifications`, beside the channels they consume.
|
||||
|
||||
Remediation introduces no new mechanism: a finding carrying `fixed_in` renders
|
||||
an **Apply updates** action calling the existing
|
||||
`POST /api/servers/:id/apply-updates`, which is already `ApplyUpdatesCmd`. See
|
||||
it, patch it, one place — and no second patching path to keep consistent with
|
||||
the first.
|
||||
|
||||
Database freshness is shown wherever findings are, not tucked into settings. A
|
||||
fleet scanning against a three-week-old database must say so rather than
|
||||
quietly report all-clear.
|
||||
|
||||
---
|
||||
|
||||
## Verification
|
||||
|
||||
**No automated tests.** The repository has none today, and by explicit
|
||||
instruction this feature adds none — no `*_test.go`, no frontend test files.
|
||||
That is a deliberate decision by the repository owner, recorded here so the
|
||||
absence reads as a choice rather than an omission.
|
||||
|
||||
It does change the risk profile, and the places it changes it are worth naming,
|
||||
because each fails by producing a **wrong answer rather than a crash**:
|
||||
|
||||
- **Version comparison.** The backport case — installed `1:3.0.2-0ubuntu1.15`
|
||||
against advisory fixed-in `1:3.0.2-0ubuntu1.15` resolving to *not
|
||||
vulnerable* — plus tilde ordering (`1.0~rc1` < `1.0`), epoch dominance
|
||||
(`1:1.0` > `2.0`) and `1.9` < `1.10`. Wrong here means a vulnerable fleet
|
||||
reported clean.
|
||||
- **Source-package fan-out.** One advisory against `openssl` must flag
|
||||
`libssl3`, `openssl` and `libssl-dev`. Matching on binary name alone silently
|
||||
finds one of three.
|
||||
- **`first_seen` preservation.** An upsert that overwrites it makes every
|
||||
finding look discovered today, and nothing surfaces that until someone reads
|
||||
a report.
|
||||
- **Fixed-before-reopen ordering.** A finding both absent from a scan and past
|
||||
its acceptance expiry must settle `fixed`, not reopen.
|
||||
|
||||
The implementation plan carries a manual verification table for each, to be
|
||||
walked before the relevant task is committed. They are the substitute for the
|
||||
tests, not a formality.
|
||||
|
||||
---
|
||||
|
||||
## Failure modes
|
||||
|
||||
| Failure | Behaviour |
|
||||
| ------- | --------- |
|
||||
| Database pull fails | Keep the last good copy and serve stale. Record `last_error`, surface `pulled_at` age. **Never clear findings** — a network blip must not read as "all fixed" |
|
||||
| Unsupported distribution | `status: unsupported`, not zero findings |
|
||||
| Agent stops reporting | Findings persist and `collected_at` age is shown. No auto-expiry: a silent agent is not a patched server |
|
||||
| `trivy-db` schema version bumps | The puller refuses an unknown schema rather than mis-parsing it |
|
||||
| ghcr anonymous rate limit | Backoff; `VANTAGE_TRIVY_DB_REF` mirrors to a private registry |
|
||||
| Leadership lost mid-scan | The context is cancelled and the scan returns; `scan_pending` is still set, so the next leader picks it up |
|
||||
| Instance deleted | **`server_packages` and `vuln_findings` must be added to the control plane's instance-deletion collection list.** Easy to miss, and missing it orphans a tenant's package data indefinitely |
|
||||
|
||||
---
|
||||
|
||||
## Environment variables
|
||||
|
||||
| Name | Required | Notes |
|
||||
| ---- | -------- | ----- |
|
||||
| `VANTAGE_TRIVY_DB_REF` | no | default `ghcr.io/aquasecurity/trivy-db:2`. Point at a mirror for air-gapped installs or to avoid the anonymous ghcr rate limit |
|
||||
| `VANTAGE_VULNDB_DISABLED` | no | disables the puller and the scheduler entirely. Findings already written are still served and still marked stale |
|
||||
|
||||
---
|
||||
|
||||
## Deliberately out of scope
|
||||
|
||||
- **Windows.** Separate source, collector and matcher; its own spec.
|
||||
- **Container image scanning.** Sub-project C; needs the container registry.
|
||||
- **Compliance baseline assertions.** Sub-project D; shares this findings UI
|
||||
and nothing else.
|
||||
- **Language-level dependency scanning** (npm, pip, Go modules). `trivy-db`
|
||||
covers these ecosystems, but finding the manifests on a host is a different
|
||||
collection problem from asking the package manager what is installed.
|
||||
- **Automatic patching on a finding.** Remediation is one click, not zero. An
|
||||
unattended upgrade triggered by a CVE feed is a fleet-wide change driven by a
|
||||
third party's data, which is not a decision to take away from an operator.
|
||||
@@ -1,441 +0,0 @@
|
||||
# Workload registry
|
||||
|
||||
Date: 2026-08-06
|
||||
|
||||
Agents enumerate what each server actually runs — Docker containers, the
|
||||
compose stacks grouping them, and systemd services — and report it to the
|
||||
control plane. Containers and units can be started, stopped and restarted from
|
||||
the UI, and a bounded snapshot of their logs can be read without opening a
|
||||
console.
|
||||
|
||||
This is **sub-project B** of the four sketched in
|
||||
`2026-08-06-package-inventory-and-cve-findings-design.md`:
|
||||
|
||||
| # | Sub-project | Depends on |
|
||||
| - | ----------- | ---------- |
|
||||
| A | Package inventory + CVE findings — its own spec | nothing |
|
||||
| B | **Workload registry** — this spec | nothing |
|
||||
| C | Container image scanning | A and B |
|
||||
| D | Compliance profiles | shares A's findings UI only |
|
||||
|
||||
A and B are independent. C is the joiner and must not be designed before both
|
||||
exist: it needs B's image list and A's findings model.
|
||||
|
||||
**Workload** is the domain word throughout: one container or one systemd unit.
|
||||
It gives the collection, the commands and the page a single honest name rather
|
||||
than saying "container or service" in every identifier.
|
||||
|
||||
Scope is **Linux only**, matching sub-project A and the existing position that
|
||||
Windows agents are second-class by design. Docker runs on Windows; systemd does
|
||||
not, and half a feature per platform is worse than a clear line.
|
||||
|
||||
---
|
||||
|
||||
## What this is for
|
||||
|
||||
The control plane can manage a fleet's keys, run workflows across it and watch
|
||||
its endpoints, but it has no idea what any of those servers actually *runs*.
|
||||
"Restart nginx on that box" means opening a console. "Which of these 80 servers
|
||||
is still on the old image" is unanswerable.
|
||||
|
||||
---
|
||||
|
||||
## Reporting and refresh are one path
|
||||
|
||||
The agent reports on its own 60-second ticker through a `ReportWorkloads` RPC,
|
||||
using the same hash short-circuit as the package report: it offers a SHA-256 of
|
||||
the sorted workload list, and sends the body only when the server does not
|
||||
already hold that hash. An unchanged list costs one small message, which on a
|
||||
60-second cadence is the common case by a wide margin.
|
||||
|
||||
The on-demand refresh **does not return data**. `RefreshWorkloadsCmd` carries
|
||||
no payload back; it makes the agent report immediately through the normal RPC,
|
||||
and the UI refetches the stored document.
|
||||
|
||||
That is deliberate. A refresh that returned workloads inline would be a second
|
||||
writer for the same collection, arriving by a different route, with its own
|
||||
serialisation and its own opportunity to disagree with the periodic one. One
|
||||
writer, one shape; the refresh is a nudge, not a channel.
|
||||
|
||||
Opening a server's Workloads tab dispatches a refresh, so what is on screen is
|
||||
live rather than up to a minute stale. That matters because the page has a
|
||||
Restart button on it: a stale list is not merely a wrong impression, it is a
|
||||
wrong action aimed at a container that already died.
|
||||
|
||||
## What does answer back
|
||||
|
||||
Two operations genuinely return something:
|
||||
|
||||
| Command | Answers with |
|
||||
| ------- | ------------ |
|
||||
| `ControlWorkloadCmd{kind, id, action}` | the existing `CommandResult` — ok or error |
|
||||
| `WorkloadLogsCmd{kind, id, tail}` | a new `WorkloadLogsResult{command_id, text, truncated}` |
|
||||
|
||||
Both ride the proven path: `commandDispatcher.send()` for request and ack, and
|
||||
a `WorkloadResults` registry mirroring `StepResults.Await`/`Deliver` over the
|
||||
bus. **`Await` must subscribe before the command is dispatched** — the pod
|
||||
driving the request is usually not the pod holding the agent's stream, and a
|
||||
fast agent otherwise answers into a channel nobody has joined. This is not a
|
||||
new hazard; it is the one `stepresults.go` already documents.
|
||||
|
||||
```protobuf
|
||||
rpc ReportWorkloads(ReportWorkloadsRequest) returns (ReportWorkloadsResponse);
|
||||
|
||||
message ReportWorkloadsRequest {
|
||||
string server_id = 1;
|
||||
string agent_token = 2;
|
||||
string hash = 3;
|
||||
bool docker_ok = 4;
|
||||
string docker_error = 5;
|
||||
bool systemd_ok = 6;
|
||||
string systemd_error = 7;
|
||||
repeated Workload workloads = 8; // empty on the offer call
|
||||
}
|
||||
|
||||
message ReportWorkloadsResponse {
|
||||
bool need_full = 1;
|
||||
}
|
||||
|
||||
// ServerCommand gains three variants.
|
||||
message RefreshWorkloadsCmd {}
|
||||
|
||||
message ControlWorkloadCmd {
|
||||
string kind = 1; // "container" | "unit"
|
||||
string id = 2;
|
||||
string action = 3; // "start" | "stop" | "restart"
|
||||
}
|
||||
|
||||
message WorkloadLogsCmd {
|
||||
string kind = 1;
|
||||
string id = 2;
|
||||
int32 tail = 3;
|
||||
}
|
||||
|
||||
// AgentMessage gains one variant.
|
||||
message WorkloadLogsResult {
|
||||
string command_id = 1;
|
||||
string text = 2;
|
||||
bool truncated = 3;
|
||||
string error = 4;
|
||||
}
|
||||
```
|
||||
|
||||
The offer-then-send handshake is the package report's, unchanged: the agent
|
||||
calls once with `workloads` empty, and resends with the body only if the
|
||||
response sets `need_full`.
|
||||
|
||||
An agent whose stream no pod holds gets a 503 from the dispatcher, as
|
||||
everything else does. Commands are not queued: a command whose owner died must
|
||||
fail loudly rather than be delivered to nobody while the operator is told it
|
||||
worked.
|
||||
|
||||
---
|
||||
|
||||
## Not gated by licence
|
||||
|
||||
Unlike CVE scanning, this reads as core fleet management rather than a premium
|
||||
add-on, so v1 ships to every instance with no entitlement check.
|
||||
|
||||
If that changes it is a one-line `HasFeature` check at `ReportWorkloads`,
|
||||
gating collection rather than display — the same placement and the same
|
||||
reasoning as sub-project A, where gating the UI alone would still pay every
|
||||
write cost.
|
||||
|
||||
---
|
||||
|
||||
## Data model
|
||||
|
||||
One new collection, `server_workloads`, one document per server, mirroring
|
||||
`server_packages`.
|
||||
|
||||
```go
|
||||
type ServerWorkloads struct {
|
||||
ID bson.ObjectID `bson:"_id,omitempty" json:"-"`
|
||||
InstanceID string `bson:"instance_id" json:"-"`
|
||||
ServerID string `bson:"server_id" json:"server_id"`
|
||||
Hash string `bson:"hash" json:"hash"`
|
||||
Workloads []Workload `bson:"workloads" json:"workloads"`
|
||||
CollectedAt time.Time `bson:"collected_at" json:"collected_at"`
|
||||
|
||||
DockerOK bool `bson:"docker_ok" json:"docker_ok"`
|
||||
DockerError string `bson:"docker_error,omitempty" json:"docker_error,omitempty"`
|
||||
SystemdOK bool `bson:"systemd_ok" json:"systemd_ok"`
|
||||
SystemdError string `bson:"systemd_error,omitempty" json:"systemd_error,omitempty"`
|
||||
}
|
||||
|
||||
type Workload struct {
|
||||
Kind string `bson:"kind" json:"kind"` // "container" | "unit"
|
||||
ID string `bson:"id" json:"id"` // container id, or unit name
|
||||
Name string `bson:"name" json:"name"`
|
||||
State string `bson:"state" json:"state"`
|
||||
Health string `bson:"health,omitempty" json:"health,omitempty"`
|
||||
Image string `bson:"image,omitempty" json:"image,omitempty"`
|
||||
Stack string `bson:"stack,omitempty" json:"stack,omitempty"`
|
||||
Ports []string `bson:"ports,omitempty" json:"ports,omitempty"`
|
||||
Restarts int `bson:"restarts,omitempty" json:"restarts,omitempty"`
|
||||
StartedAt time.Time `bson:"started_at,omitempty" json:"started_at,omitempty"`
|
||||
Protected bool `bson:"protected" json:"protected"`
|
||||
}
|
||||
```
|
||||
|
||||
`State` is normalised across the two kinds: containers report `running`,
|
||||
`exited`, `paused`, `restarting`, `created`; units report `active`, `inactive`,
|
||||
`failed`, `activating`. They are deliberately **not** collapsed into a shared
|
||||
vocabulary — a failed unit and an exited container mean different things, and
|
||||
flattening them would lose the distinction the operator needs.
|
||||
|
||||
Indexes: `{instance_id, server_id}` unique, plus multikey
|
||||
`{instance_id, "workloads.image"}` for the fleet-wide "which servers run image
|
||||
X" query.
|
||||
|
||||
### Why the OK/Error pairs exist
|
||||
|
||||
A host with no Docker installed and a host where Docker is installed and
|
||||
running nothing both produce an empty list. One should read "not in use here",
|
||||
the other "nothing running", and only the second deserves any alarm.
|
||||
|
||||
The error strings separate a third case the booleans alone cannot: Docker
|
||||
installed with the daemon down. "Not installed" and "installed but not
|
||||
responding" are different problems with different fixes, and collapsing them
|
||||
into one false boolean throws away the only thing that tells them apart.
|
||||
|
||||
### Why `Protected` is reported rather than derived
|
||||
|
||||
The agent already knows which unit and container it is. Sending that up lets
|
||||
the UI render the action disabled with a reason instead of offering a button
|
||||
whose refusal is already known.
|
||||
|
||||
The field is the courtesy; the agent's own check is the boundary. See the
|
||||
control section.
|
||||
|
||||
### No history
|
||||
|
||||
A workload list is state, not a record. Nobody asks what containers ran last
|
||||
Tuesday, and keeping it would grow a collection per server per minute in
|
||||
exchange for a question nobody has.
|
||||
|
||||
---
|
||||
|
||||
## Collectors
|
||||
|
||||
### Docker: two commands, no English parsing
|
||||
|
||||
```
|
||||
docker ps -aq
|
||||
docker inspect --format '{{json .}}' <ids…>
|
||||
```
|
||||
|
||||
Not `docker ps --format '{{json .}}'` alone. That reports health and uptime
|
||||
inside a human `Status` string — `"Up 2 hours (healthy)"` — and anything built
|
||||
on it is parsing English that is localised, reworded between releases, and
|
||||
silently different for a paused or restarting container. `inspect` returns
|
||||
`State.Health.Status`, `State.StartedAt` and `RestartCount` as typed fields.
|
||||
Two execs instead of one, and no parser to be wrong.
|
||||
|
||||
`RestartCount` justifies the second call by itself: a container cycling is the
|
||||
single thing this page most needs to show, and it is invisible in a list that
|
||||
only ever says "Up".
|
||||
|
||||
Compose stacks come from the `com.docker.compose.project` label. **No YAML is
|
||||
read from disk** — the label is what Docker itself treats as authoritative, and
|
||||
a compose file on disk may not be what is actually running.
|
||||
|
||||
Docker absent, or a socket that cannot be reached, sets `DockerOK: false`. It
|
||||
is not an error and produces no log line: most servers in a fleet built around
|
||||
SSH key management will not have Docker, and treating the normal case as a
|
||||
fault makes the feature look broken on the majority of the estate.
|
||||
|
||||
### systemd: filtered on purpose
|
||||
|
||||
```
|
||||
systemctl list-units --type=service --state=running,failed --no-legend --plain --no-pager
|
||||
systemctl list-unit-files --type=service --state=enabled --no-legend --plain --no-pager
|
||||
```
|
||||
|
||||
Two calls because "running or failed" and "enabled but stopped" are different
|
||||
questions, and an enabled unit that is not running is exactly the one worth
|
||||
seeing.
|
||||
|
||||
Excluded by prefix: `systemd-`, `user@`, `session-`, `init.scope`. A typical
|
||||
host carries 300+ units, the platform's own accounting for most of them.
|
||||
Listing all of them buries the ten anyone cares about — the same failure mode
|
||||
as an unfiltered vulnerability report, and the same fix.
|
||||
|
||||
Column output rather than `--output=json`: the JSON flag requires systemd 246+,
|
||||
and this fleet includes older stable distributions. The column format has been
|
||||
stable considerably longer than the JSON one has existed.
|
||||
|
||||
---
|
||||
|
||||
## Control actions
|
||||
|
||||
```
|
||||
container: docker {start|stop|restart} <id>
|
||||
unit: systemctl {start|stop|restart} <unit>
|
||||
```
|
||||
|
||||
Owner or admin only. Every action writes an audit event naming the actor, the
|
||||
server and the target.
|
||||
|
||||
### The protected set
|
||||
|
||||
Computed agent-side: `vantage-agent.service`, plus the container ID read from
|
||||
`/proc/self/cgroup` should the agent ever be run inside a container.
|
||||
|
||||
The agent refuses those before doing anything. As with the console relay
|
||||
hardcoding `127.0.0.1` agent-side, **the control plane may name a target, but
|
||||
the agent decides what it will do to itself**. A server-side denylist alone
|
||||
would be bypassed by the next dispatch path someone adds, and the failure is
|
||||
unrecoverable from the UI: a server that stops its own agent goes offline, and
|
||||
the way back is SSH or physical access — precisely what this feature exists to
|
||||
avoid needing.
|
||||
|
||||
### Timeouts
|
||||
|
||||
`docker stop` waits on a container that may ignore SIGTERM. `systemctl stop`
|
||||
on a unit with a long `TimeoutStopSec` blocks for exactly as long as that says.
|
||||
Both run under a 90-second context, and a timeout returns a real error rather
|
||||
than an ack implying success.
|
||||
|
||||
---
|
||||
|
||||
## Logs
|
||||
|
||||
```
|
||||
container: docker logs --tail 500 --timestamps <id>
|
||||
unit: journalctl -u <unit> -n 500 --no-pager --output=short-iso
|
||||
```
|
||||
|
||||
Capped at **500 lines and 256KB, whichever binds first**, with `truncated` set
|
||||
so the UI can say so. Two caps because 500 lines of a container emitting 4KB
|
||||
JSON blobs is 2MB, and a line count alone does not stop it — the same reasoning
|
||||
that gave workflow logs both a per-line and a per-run cap.
|
||||
|
||||
Live following is deliberately absent. The browser console already offers a
|
||||
real terminal on the same server, where `docker logs -f` works properly with
|
||||
its own scrollback and cancellation. Building a second streaming path — a
|
||||
relay listener, proxy bus keys, a WebSocket upgrade and a cancellation story
|
||||
for a follow nobody closed — to duplicate that would be a large amount of
|
||||
machinery aimed at a capability already shipped. A bounded snapshot answers
|
||||
"why did this restart", which is the question that sends people to the console
|
||||
in the first place.
|
||||
|
||||
### Log reads are owner or admin only, and audited
|
||||
|
||||
Unlike workflow logs, these cannot be masked. A workflow's logs can be masked
|
||||
because the run injected the secrets and therefore knows their values. A
|
||||
container's stdout is arbitrary and may contain credentials nobody declared —
|
||||
a connection string in a startup banner, a token in a stack trace.
|
||||
|
||||
So log reads sit behind the same role check as control actions and are audited.
|
||||
A member who can see the fleet cannot read its logs. This is a deliberate
|
||||
access decision, not an oversight, and it is why log reading is not simply
|
||||
folded in with the read-only snapshot endpoints.
|
||||
|
||||
---
|
||||
|
||||
## REST API
|
||||
|
||||
```
|
||||
GET /api/servers/:id/workloads # stored snapshot
|
||||
POST /api/servers/:id/workloads/refresh # dispatch, then refetch
|
||||
POST /api/servers/:id/workloads/:wid/action # {"action":"start|stop|restart"} (owner|admin)
|
||||
GET /api/servers/:id/workloads/:wid/logs?tail= # (owner|admin)
|
||||
GET /api/workloads?image=&stack=&state= # fleet-wide
|
||||
```
|
||||
|
||||
`:wid` is a container ID or a unit name, URL-encoded. Unit names carry dots and
|
||||
`@`, which are legal in a path segment but not worth relying on unencoded.
|
||||
|
||||
`tail` is clamped to the 500-line cap server-side; a client asking for more
|
||||
gets 500, not an error.
|
||||
|
||||
---
|
||||
|
||||
## UI
|
||||
|
||||
Server detail gains a **Workloads** tab, ordered compose stacks first — grouped
|
||||
under the stack name — then loose containers, then units.
|
||||
|
||||
That ordering is not cosmetic. A stack is one thing to an operator even when it
|
||||
is six containers, and a flat list turns one decision into six rows. It is the
|
||||
same argument that groups the vulnerabilities board by CVE rather than by
|
||||
finding.
|
||||
|
||||
A `/workloads` fleet view answers "which servers run image X", which is the
|
||||
reason the snapshot is stored at all rather than fetched on demand and
|
||||
discarded.
|
||||
|
||||
Three rules that follow directly from the model:
|
||||
|
||||
- **Protected rows render their actions disabled, with the reason**, rather
|
||||
than offering a button whose refusal is already known.
|
||||
- **`DockerOK: false` reads "Docker not in use on this server"**, never an
|
||||
empty list, and `DockerError` when present is shown as a distinct problem.
|
||||
- State never reads by colour alone: every pill carries a distinct shape and a
|
||||
text label, matching the existing monitor and severity pills.
|
||||
|
||||
---
|
||||
|
||||
## Verification
|
||||
|
||||
**No automated tests.** The repository has none today, and by explicit
|
||||
instruction this feature adds none — no `*_test.go`, no frontend test files.
|
||||
A deliberate decision by the repository owner, recorded so the absence reads as
|
||||
a choice rather than an omission.
|
||||
|
||||
The behaviours that would otherwise have been tested are the ones that fail
|
||||
quietly, and the implementation plan carries a manual check for each:
|
||||
|
||||
- **Parser output against real command output.** `docker inspect` must yield
|
||||
`RestartCount`, health and the compose label as `Stack`; the `systemctl`
|
||||
exclusion filter must drop `systemd-*` and `user@*` while keeping
|
||||
`nginx.service`. Both are verified against a live host rather than a fixture.
|
||||
- **Protected-set computation.** `vantage-agent.service` marked,
|
||||
`nginx.service` not. Getting this wrong in the permissive direction lets a
|
||||
server stop its own agent, which is unrecoverable from the UI.
|
||||
- **Hash order-independence.** An ordering-sensitive hash resends the full list
|
||||
every 60 seconds, which is invisible except as traffic.
|
||||
- **Log capping in both directions.** 600 lines in → 500 out with `truncated`;
|
||||
a 300KB blob of fewer than 500 lines → capped, `truncated`. The second is the
|
||||
case a line-count-only implementation silently fails, and it fails by sending
|
||||
megabytes rather than by erroring.
|
||||
|
||||
---
|
||||
|
||||
## Failure modes
|
||||
|
||||
| Failure | Behaviour |
|
||||
| ------- | --------- |
|
||||
| Docker not installed | `DockerOK: false`, no error, UI reads "not in use" |
|
||||
| Docker installed, daemon down | `DockerOK: false` **plus** `DockerError` — different message, different fix |
|
||||
| Agent offline | 503 from the existing dispatcher. No queueing: a command whose owner died must fail loudly |
|
||||
| Action on a protected workload | Agent refuses; API answers 409 naming the reason |
|
||||
| `stop` exceeds its timeout | Real error surfaced, never a hopeful ack. Snapshot refreshed afterwards |
|
||||
| Container removed between snapshot and action | Docker's "No such container" surfaced and a refresh dispatched — this is what on-demand refresh is for |
|
||||
| Log exceeds either cap | Truncated, flagged, and stated in the UI |
|
||||
| Instance deleted | **`server_workloads` must be added to the control plane's instance-deletion collection list**, alongside sub-project A's two collections |
|
||||
|
||||
---
|
||||
|
||||
## Deliberately out of scope
|
||||
|
||||
- **Live log following.** The console already does it. See the logs section.
|
||||
- **Creating, deleting or updating containers and units.** This is a control
|
||||
and visibility surface, not a deployment tool — workflows already exist for
|
||||
changing what a server runs, with snapshots, audit and rollback.
|
||||
- **`docker exec` into a container.** The console reaches the host; exec from
|
||||
the control plane is a second remote-execution path with its own audit and
|
||||
authorisation story, and it belongs in its own spec if anywhere.
|
||||
- **Kubernetes and containerd.** The Docker collector shells to the `docker`
|
||||
CLI, so a node whose runtime is containerd or CRI-O reports nothing from it —
|
||||
`DockerOK: false`, correctly, since Docker genuinely is not in use. Covering
|
||||
those runtimes means a `crictl`/`nerdctl` collector, and talking to a
|
||||
Kubernetes API server is a different subsystem again. Neither is v1.
|
||||
- **Podman as a supported runtime.** Its `docker`-compatible CLI means an
|
||||
aliased install will largely work, and that is a happy accident rather than a
|
||||
claim: nothing here is tested against Podman and its `RestartCount` and
|
||||
compose-label behaviour are not verified.
|
||||
- **Windows.** No systemd, and a different container story.
|
||||
- **Image vulnerability scanning.** Sub-project C, which needs this spec's
|
||||
image list and sub-project A's findings model.
|
||||
@@ -1,297 +0,0 @@
|
||||
# API tokens and OpenAPI reference
|
||||
|
||||
Date: 2026-08-12
|
||||
Status: approved, ready for implementation planning
|
||||
|
||||
## Problem
|
||||
|
||||
The only programmatic credential the control plane issues is the ESO secrets-read
|
||||
bearer token, which reaches exactly one endpoint. Everything else requires a
|
||||
browser session cookie. There is therefore no supported way to drive Vantage from
|
||||
CI, a script, or infrastructure-as-code, and no machine-readable description of
|
||||
the REST API for anyone who wants to try.
|
||||
|
||||
This spec covers two deliverables that ship together: scoped API tokens, and an
|
||||
OpenAPI 3.1 document rendered as a live reference page. A Terraform provider is
|
||||
the intended follow-on and is explicitly out of scope here — it depends on both
|
||||
of these being settled, and it is a separate Go module with its own release
|
||||
cycle.
|
||||
|
||||
## Goals
|
||||
|
||||
- A person can mint a scoped, optionally expiring token and use it against the
|
||||
existing REST API with no new endpoints to learn.
|
||||
- A leaked token is bounded by role, by scope, and by expiry policy.
|
||||
- Offboarding a person removes their tokens as a side effect of removing them.
|
||||
- The API has a machine-readable description that cannot silently drift from the
|
||||
handlers it describes.
|
||||
- The reference page works on an air-gapped self-hosted install.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Token editing. Role and scopes are immutable; rotation replaces amendment.
|
||||
- OAuth device flow or any browser-based authorisation grant.
|
||||
- Per-server or per-tag restrictions on a token.
|
||||
- Instance-owned service tokens that outlive their creator.
|
||||
- The Terraform provider.
|
||||
- General API rate limiting beyond the per-token limit described below.
|
||||
|
||||
## Part 1 — API tokens
|
||||
|
||||
### Token format and storage
|
||||
|
||||
A token is `vt_` followed by 32 random bytes, hex encoded. It is displayed once,
|
||||
at creation, and never again.
|
||||
|
||||
Only the SHA-256 hash is stored, in a unique index. This follows the precedent
|
||||
already set by `servers.agent_token_hash` and the ESO read token. bcrypt is
|
||||
deliberately not used: the value is full-entropy random rather than a
|
||||
user-chosen password, so a fast hash is sufficient, and a per-token salt would
|
||||
force a collection scan where an indexed lookup is wanted.
|
||||
|
||||
The first eight characters are stored in clear as `hint`, so the list can
|
||||
identify a token without revealing it.
|
||||
|
||||
### Authentication path
|
||||
|
||||
`auth.Middleware()` gains a fallback. When there is no `km_session` cookie it
|
||||
looks for `Authorization: Bearer vt_…`. Both paths end by placing a `*Session` in
|
||||
the gin context, so every handler, `auth.RequireRole`, `RequireActiveLicense`,
|
||||
`RequireFeature` and `actorFromCtx` continue to work unmodified.
|
||||
|
||||
```
|
||||
Session{
|
||||
UserID: token.UserID
|
||||
InstanceID: token.InstanceID
|
||||
Role: min(user.Role, token.Role) // owner > admin > member
|
||||
Email: user.Email
|
||||
TokenID: token.TokenID // "" for cookie sessions
|
||||
Scopes: token.Scopes // nil for cookie sessions
|
||||
}
|
||||
```
|
||||
|
||||
The effective role is recomputed on every request rather than frozen at
|
||||
creation. Demoting the user demotes the token with them. No caching is required
|
||||
because the user document is already read to confirm the user still exists.
|
||||
|
||||
The existing host guard applies identically. A token carries an `instance_id`,
|
||||
and a request arriving at a different instance's host is rejected exactly as a
|
||||
mismatched cookie session is. The tenant boundary must not have a token-shaped
|
||||
hole in it.
|
||||
|
||||
`last_used_at` is written best-effort and only when the stored value is more
|
||||
than 60 seconds old, so it does not become a Mongo write per request.
|
||||
|
||||
Rejections:
|
||||
|
||||
| Condition | Status | Body |
|
||||
| -------------------- | ------ | -------------------------------------- |
|
||||
| No credential at all | 401 | `not authenticated` |
|
||||
| Unknown token | 401 | `invalid token` |
|
||||
| Expired token | 401 | `code: token_expired` |
|
||||
| Owning user deleted | 401 | `invalid token` |
|
||||
| Missing scope | 403 | names the required scope |
|
||||
| Wrong instance host | 403 | `instance host mismatch` |
|
||||
|
||||
### Data model
|
||||
|
||||
New collection `api_tokens`, added to `services.ScopedCollections` so instance
|
||||
purge reaches it.
|
||||
|
||||
```
|
||||
instance_id string
|
||||
token_id string
|
||||
user_id string
|
||||
name string 1-64 chars, unique per user
|
||||
hint string first 8 chars of the plaintext
|
||||
token_hash string sha256
|
||||
role string owner|admin|member
|
||||
scopes []string
|
||||
expires_at *time.Time nil means never
|
||||
created_at time.Time
|
||||
last_used_at *time.Time
|
||||
created_by_ip string
|
||||
```
|
||||
|
||||
Indexes: unique on `token_hash`; compound on `(instance_id, user_id)`.
|
||||
|
||||
Deleting a user deletes their tokens as part of the same service call as
|
||||
`DeleteInstanceUser`, so offboarding is one action rather than two.
|
||||
|
||||
### Expiry policy
|
||||
|
||||
Expiry is optional by default: a token may be created with no expiry at all.
|
||||
Instance settings gain `api_token_max_days *int`, editable by owner and admin:
|
||||
|
||||
- `nil` — no cap; never-expire is allowed. This is the default, so an upgrade
|
||||
changes nothing.
|
||||
- `n > 0` — a new token must expire within `n` days, and a never-expire token is
|
||||
refused.
|
||||
|
||||
Changing the setting does not retroactively invalidate existing tokens; it is a
|
||||
policy on issuance. Tokens already outside the new cap are flagged in the UI so
|
||||
that someone can rotate them deliberately, rather than discovering the change
|
||||
when a pipeline breaks.
|
||||
|
||||
### Scopes
|
||||
|
||||
Eight resources, each with `:read` and `:write`. Write implies read on the same
|
||||
resource.
|
||||
|
||||
```
|
||||
servers keys secrets workflows
|
||||
monitors vulns workloads settings
|
||||
```
|
||||
|
||||
Scope enforcement is a single middleware, `RequireScopes()`, mounted once in the
|
||||
`/api` stack. It derives the required resource from the matched gin route
|
||||
pattern using a map, rather than from a per-route decorator: a route registered
|
||||
without a decorator would otherwise be unguarded, and this repo already prefers
|
||||
guards that come from where a route is mounted rather than from someone
|
||||
remembering.
|
||||
|
||||
- Cookie sessions skip the check entirely.
|
||||
- A token-authenticated request whose route pattern is absent from the map is
|
||||
denied with 403. Fail closed.
|
||||
- A startup check fails boot if any registered `/api` route pattern is missing
|
||||
from the map, so the failure surfaces at deploy rather than at the first call.
|
||||
|
||||
Deliberate placements:
|
||||
|
||||
- `keys:read` covers `GET /keys/:id/private-key`. Reading a private key is
|
||||
reading a key.
|
||||
- `secrets:read` does not cover `GET /api/secrets/:group/values`. That endpoint
|
||||
keeps its separate ESO bearer path and is unaffected by this work.
|
||||
- `workloads:write` covers both container control actions and log reads, which
|
||||
are already restricted to owner and admin.
|
||||
- The token endpoints themselves map to the `settings` resource: `GET
|
||||
/api/tokens` requires `settings:read`, and `POST` and `DELETE` require
|
||||
`settings:write`. A token can therefore mint or revoke tokens only when
|
||||
explicitly granted that scope, and never above its own role.
|
||||
|
||||
### Endpoints
|
||||
|
||||
```
|
||||
GET /api/tokens list; a member sees their own, owner|admin see all
|
||||
POST /api/tokens create; returns the plaintext once
|
||||
DELETE /api/tokens/:id revoke; own always, owner|admin any
|
||||
```
|
||||
|
||||
There is no `PUT`. Editing a token's role or scopes changes what a credential
|
||||
already deployed in a CI system can do, with no record of what it could do
|
||||
before. Rotation replaces amendment.
|
||||
|
||||
`POST` body: `name`, `role`, `scopes[]`, `expires_in_days` (omitted means never,
|
||||
and is refused when `api_token_max_days` is set).
|
||||
|
||||
Refusals: 400 for an unknown scope, 409 for a duplicate name for that user, 403
|
||||
for a role above the creator's own, 422 for an expiry beyond policy.
|
||||
|
||||
### Web UI
|
||||
|
||||
A new "API tokens" card in the Access group of `/settings`, alongside Members
|
||||
and single sign-on. Not a new nav entry — `/settings/instance` was folded back
|
||||
into `/settings` for precisely this reason, and the card lives in
|
||||
`web/components/settings/` with the others, reusing the shared `Field` and
|
||||
`inputClass`.
|
||||
|
||||
The card lists name, hint, role, scope chips, last used, and expiry with a
|
||||
distinct state for expired and for over-policy. Revoke is per row and confirms.
|
||||
|
||||
Create opens a modal. The plaintext is shown once in a `--well` block with
|
||||
copy-to-clipboard and an explicit line saying it will not be shown again.
|
||||
|
||||
Members see only their own rows. Owner and admin get an "All tokens" toggle.
|
||||
|
||||
`api_token_max_days` is a field on the same card, visible to owner and admin
|
||||
only.
|
||||
|
||||
### Audit
|
||||
|
||||
New events:
|
||||
|
||||
- `token.created`
|
||||
- `token.revoked`
|
||||
- `token.expired_use` — a rejected expired token, which is how a forgotten CI
|
||||
job becomes visible
|
||||
- `settings.token_policy_updated`
|
||||
|
||||
The actor is the human's email throughout, so `actorFromCtx` needs no change.
|
||||
Every existing audit event written during a token-authenticated request gains
|
||||
`via: "token:<name>"` in its detail, so the log distinguishes a person clicking
|
||||
from their credential acting.
|
||||
|
||||
### Rate limiting
|
||||
|
||||
Token-authenticated requests are limited per token in Redis at 600 per minute,
|
||||
answering 429 with `Retry-After`. Cookie sessions are untouched. This is narrow
|
||||
on purpose: it is not the general API rate-limiting project, only enough that a
|
||||
runaway script cannot take an instance down.
|
||||
|
||||
## Part 2 — OpenAPI and the reference page
|
||||
|
||||
### Generation
|
||||
|
||||
`swaggo/swag` v2, pinned, emitting OpenAPI 3.1. v1 emits Swagger 2.0, which
|
||||
Scalar renders poorly.
|
||||
|
||||
Handlers in `server/internal/api/*.go` gain annotation comments. Request and
|
||||
response bodies that are currently anonymous inline structs become named
|
||||
structs. This is real churn across roughly fifteen files and is the honest cost
|
||||
of choosing generation over a hand-written document.
|
||||
|
||||
The generated `server/internal/api/docs/openapi.json` is committed and embedded
|
||||
with `go:embed`, not generated during the image build: `server/Dockerfile`
|
||||
produces a `scratch` runtime from a Go build stage, and adding codegen there
|
||||
means putting the toolchain in the build image.
|
||||
|
||||
`server-deploy.yml` gains a check that regenerates the spec and runs
|
||||
`git diff --exit-code`. An annotation edited without regenerating fails the
|
||||
build. Without this check the annotations are worth less than a hand-written
|
||||
document, because they would drift while appearing authoritative.
|
||||
|
||||
### Serving
|
||||
|
||||
```
|
||||
GET /api/openapi.json the spec, session or token authenticated
|
||||
GET /api/docs HTML page loading a vendored Scalar bundle
|
||||
```
|
||||
|
||||
The Scalar standalone bundle is vendored under `server/internal/api/docs/`, with
|
||||
its version recorded in a comment beside it and refreshed by hand. No CDN:
|
||||
air-gapped self-hosted installs are supported, and a reference page that fails
|
||||
closed on an offline site is a support ticket.
|
||||
|
||||
Because the page is served by the instance itself, "Try it" acts against the
|
||||
reader's own API with their own session.
|
||||
|
||||
### Documented auth schemes
|
||||
|
||||
Three, kept distinct:
|
||||
|
||||
- `cookieAuth` — the `km_session` cookie.
|
||||
- `bearerAuth` — a `vt_…` API token.
|
||||
- The ESO secrets endpoint is marked as its own separate scheme, so nobody wires
|
||||
a personal access token into External Secrets Operator.
|
||||
|
||||
## Documentation
|
||||
|
||||
- `docsite/docs/reference/api-tokens.md`: creating a token, the scope table,
|
||||
curl examples, rotation, and the maximum-lifetime policy.
|
||||
- `CLAUDE.md`: the three token routes under REST API, the `api_tokens`
|
||||
collection, and a note that `openapi.json` is generated and CI-verified.
|
||||
|
||||
## Risks
|
||||
|
||||
- The anonymous-struct-to-named-struct conversion is the bulk of the work and
|
||||
touches handler code this feature otherwise has no business in.
|
||||
- The vendored Scalar bundle is a manual refresh that nobody will remember. The
|
||||
version comment is the only mitigation.
|
||||
- A scope map keyed on gin route patterns breaks if a route path is renamed. The
|
||||
boot-time completeness check is what turns that into a startup failure rather
|
||||
than a silent 403 in production.
|
||||
|
||||
## Follow-on work
|
||||
|
||||
A Terraform provider, as its own spec and plan, consuming the tokens and the
|
||||
OpenAPI document produced here.
|
||||
@@ -1,236 +0,0 @@
|
||||
# Instance rename in Vantage HQ
|
||||
|
||||
**Date:** 2026-08-12
|
||||
**Status:** approved, not yet implemented
|
||||
|
||||
## Problem
|
||||
|
||||
A cloud instance is named once, at creation, and never again. The name is
|
||||
chosen in the first thirty seconds of a customer's relationship with the
|
||||
product — before they have decided whether this is "Acme" or "Acme
|
||||
Production" — and it is the name that becomes their DNS host, appears in every
|
||||
sign-in link and heads every page of their control plane. Today the only way to
|
||||
change it is to create a second instance and move, or to open a support ticket
|
||||
that has no tooling behind it.
|
||||
|
||||
## What a rename is
|
||||
|
||||
One customer-initiated action on a **cloud** instance: a new name, from which a
|
||||
new slug is derived, which moves the instance to a new DNS host.
|
||||
|
||||
Name and slug move together. The slug is re-derived through
|
||||
`provision.BaseSlug`, so the rules that named the instance at creation are the
|
||||
rules that rename it — the same reserved-label list, the same 3–40 character
|
||||
bound, the same `Slugify` collapse of non-alphanumeric runs. There is no
|
||||
separate slug field for the customer to edit, because two fields invite the
|
||||
state where the name says one thing and the host says another, and that
|
||||
divergence is exactly what a rename exists to fix.
|
||||
|
||||
A licence binds an instance **UUID**, not a slug. A rename therefore issues no
|
||||
licence, calls Paddle not at all, and consumes no relink. This is the property
|
||||
that makes the whole feature cheap, and it should be stated in any future change
|
||||
that tempts someone to touch the licence from this path.
|
||||
|
||||
### What breaks, deliberately
|
||||
|
||||
- **The old host stops working.** The old slug is released the moment the rename
|
||||
commits; another account may take it. Bookmarks, saved sign-in links and any
|
||||
agent install one-liner that named the web host are stale. Agents themselves
|
||||
are unaffected — they dial `GRPC_HOST`, which is not per-tenant.
|
||||
- **The old host keeps working for up to 60 seconds.** `server/internal/auth/instancehost.go`
|
||||
caches slug-to-instance lookups for 60s, and admin has no path to invalidate
|
||||
another process's memory. The released slug can be claimed by another account
|
||||
inside that window, so for up to a minute a replica still maps that host to the
|
||||
previous tenant. No data is exposed — the host/session guard rejects a session
|
||||
belonging to a different instance — but the new owner's users can briefly reach
|
||||
the old tenant's instance on their own host, and see its login page rather than
|
||||
theirs. Adding a cross-service invalidation channel for a 60-second window is
|
||||
not worth the coupling.
|
||||
- **The customer must sign in again.** `km_session` is set with no `Domain`
|
||||
attribute, so it is host-only and does not follow the instance to its new
|
||||
subdomain. The UI says so rather than letting the customer discover it.
|
||||
|
||||
## Scope
|
||||
|
||||
| | Customer (owner or admin) | Staff |
|
||||
|---|---|---|
|
||||
| Cloud instance | rename, 24h cooldown | rename, no cooldown |
|
||||
| Self-hosted instance | refused, 400 | name only; there is no slug |
|
||||
| Cloud placeholder | refused, 409 | refused, 409 |
|
||||
|
||||
Self-hosted is refused on the customer side for the same reason the member
|
||||
endpoints refuse it: there is no control-plane row to write. The install is the
|
||||
customer's, on their own database, and admin cannot reach it. Staff may still
|
||||
correct the label on admin's own row, because that label is what staff search
|
||||
by.
|
||||
|
||||
## Data flow
|
||||
|
||||
Two writes, in this order:
|
||||
|
||||
1. **Control plane `instances`** — `{name, slug}`.
|
||||
2. **Admin `admin_instances`** — `{name, slug, renamed_at}`.
|
||||
|
||||
The control plane goes first because `instances.slug` carries the unique index,
|
||||
and that index is what actually decides a race between two accounts reaching for
|
||||
the same name. Deciding it anywhere else would be guessing.
|
||||
|
||||
If the second write fails, the first is rolled back best-effort — restoring the
|
||||
previous name and slug — and the request answers 500. Leaving them divergent
|
||||
would have HQ print a host that is not the host, which is worse than a failed
|
||||
rename.
|
||||
|
||||
## Backend
|
||||
|
||||
### `shared/provision/instance.go`
|
||||
|
||||
```go
|
||||
// ErrSlugTaken means the derived slug belongs to another instance.
|
||||
var ErrSlugTaken = errors.New("slug taken")
|
||||
|
||||
// RenameInstance changes an instance's name and re-derives its slug.
|
||||
func RenameInstance(ctx context.Context, db *mongo.Database, instanceID, name string) (*models.Instance, error)
|
||||
```
|
||||
|
||||
It lives beside `CreateInstanceWithID` so slug derivation keeps one home, and it
|
||||
behaves as that function's rules imply:
|
||||
|
||||
- `BaseSlug(name)` failures wrap `ErrNameRejected` — too short, too long,
|
||||
reserved.
|
||||
- The derived slug is compared against the instance's current one. If they are
|
||||
equal, only the name is written; a cosmetic capitalisation change is not a
|
||||
move, and must not fail on its own slug.
|
||||
- **No `-2` suffix loop.** Creation appends a counter because the customer is
|
||||
waiting on an instance and any free slug will do. A rename is a request for a
|
||||
specific host, and silently landing the customer on `acme-2` is a worse answer
|
||||
than refusing.
|
||||
- A duplicate-key error on the update surfaces as `ErrSlugTaken`, exactly as the
|
||||
create path treats it as "that slug is taken". The pre-check is a courtesy;
|
||||
the index is the boundary.
|
||||
|
||||
### `admin/internal/cloudprov`
|
||||
|
||||
```go
|
||||
func RenameInstance(ctx context.Context, instanceID, name string) (*sharedmodels.Instance, error)
|
||||
```
|
||||
|
||||
A thin wrapper over `provision.RenameInstance` on `db.ControlDB()`. It writes
|
||||
`instances` and nothing else, so admin's documented control-plane write boundary
|
||||
— `instances` and `users`, from `cloudprov` and `inject` only — is unchanged.
|
||||
|
||||
### `admin/internal/models`
|
||||
|
||||
`Instance` gains:
|
||||
|
||||
```go
|
||||
// RenamedAt is when this instance last changed name, and backs the 24h
|
||||
// customer cooldown. The cooldown is admin's policy, so it lives on admin's
|
||||
// row rather than in the control plane, which has no opinion about how often
|
||||
// a customer may move.
|
||||
RenamedAt *time.Time `bson:"renamed_at,omitempty" json:"renamed_at,omitempty"`
|
||||
```
|
||||
|
||||
A pointer because absent means "never renamed", and a zero `time.Time` would
|
||||
read as 1 January year 1 — far enough in the past that the cooldown is inert,
|
||||
but only by accident.
|
||||
|
||||
### `PUT /api/instances/:id/name` (customer)
|
||||
|
||||
Mounted in the `cust` group behind `auth.RequireAccountRole(owner, admin)`, and
|
||||
resolving the instance through `ownedInstance` like every other instance route,
|
||||
so another account's instance answers 404 rather than 403.
|
||||
|
||||
Body: `{"name": "..."}`, trimmed before use.
|
||||
|
||||
Refusals, in the order checked:
|
||||
|
||||
| Condition | Status | Body |
|
||||
|---|---|---|
|
||||
| `deployment != cloud` | 400 | `selfHostedRefusal`, the same constant and status the member endpoints already answer with |
|
||||
| `placeholder` | 409 | instance is not provisioned yet |
|
||||
| within 24h of `renamed_at` | 429 | includes the UTC time it unlocks |
|
||||
| `provision.ErrNameRejected` | 422 | the wrapped reason, verbatim |
|
||||
| `provision.ErrSlugTaken` | 409 | that name is already in use |
|
||||
|
||||
Success returns `{"instance_id", "name", "slug", "login_url"}` and writes an
|
||||
audit entry `instance.renamed` with detail `<old-slug> -> <new-slug>`, so the
|
||||
history of a host is answerable from the audit log alone.
|
||||
|
||||
`login_url` comes from the existing `loginURLFor(slug)`, which fills `{slug}`
|
||||
into `APP_LOGIN_URL` — the same builder the licence emails already use, rather
|
||||
than a second opinion about how a tenant host is spelled. It is empty when
|
||||
`APP_LOGIN_URL` is unset, and the portal then falls back to the host string it
|
||||
already composes from the slug in `InstanceRecord` and the instance page.
|
||||
|
||||
### `PUT /api/staff/instances/:id/name`
|
||||
|
||||
The same core, without the cooldown, actor recorded as the staff user. On a
|
||||
self-hosted instance it updates `admin_instances.name` only and does not call
|
||||
`cloudprov`.
|
||||
|
||||
## Frontend (`adminsite`)
|
||||
|
||||
### `lib/slug.ts`
|
||||
|
||||
A TypeScript mirror of `provision.Slugify` and the length/reserved checks, used
|
||||
only to preview the resulting host while the customer types. It carries the same
|
||||
warning as `web/lib/targets.ts`: it is a second implementation and must change in
|
||||
the same commit as the Go one. The preview can disagree with the server — the
|
||||
409 is the answer that counts.
|
||||
|
||||
### `components/RenamePanel.tsx`
|
||||
|
||||
An inline panel, not a modal — `adminsite` has no modal component, and the
|
||||
codebase's idiom for a destructive-ish action with one input is `RelinkPanel`:
|
||||
a control that expands in place inside a `Panel`.
|
||||
|
||||
Prefilled with the current name. Below the input, a live line reading
|
||||
`acme-ltd.vantage.hostxtra.co.uk` as the customer types, and a note that they
|
||||
will need to sign in again on the new host. Submit is disabled while the derived
|
||||
slug is unchanged or invalid.
|
||||
|
||||
It lives in an "Address" panel on `app/(customer)/instances/[id]/page.tsx`,
|
||||
rendered only when the instance is cloud and `account_role` is `owner` or
|
||||
`admin`. The staff instance page mounts the same component against the staff
|
||||
route.
|
||||
|
||||
`InstanceRecord` on the Overview page is not touched: it stays a summary, and
|
||||
the rename is a decision that deserves the detail page.
|
||||
|
||||
### After a successful rename
|
||||
|
||||
Invalidate `["account"]`, collapse the panel, and let the page redraw with the new
|
||||
name and host. The Console rail card shows the new host, with a note:
|
||||
|
||||
> This instance now lives at `acme-ltd.vantage.hostxtra.co.uk`. You will need to
|
||||
> sign in again there.
|
||||
|
||||
**No automatic redirect.** Sending the browser to the new host lands the customer
|
||||
on a login screen with no explanation, having just lost the HQ page they were
|
||||
standing on. The link is right there; they click it when they are ready.
|
||||
|
||||
## Verification
|
||||
|
||||
The repository has no Go test suite, so verification is build plus manual
|
||||
exercise, matching existing practice:
|
||||
|
||||
- `go build ./...` in `shared` and `admin`; `npm run build` in `adminsite`.
|
||||
- Rename a cloud instance; confirm `instances` and `admin_instances` agree on
|
||||
name and slug.
|
||||
- The new host serves a login page; the old host stops resolving to the instance
|
||||
within ~60 seconds.
|
||||
- A second rename within 24 hours answers 429.
|
||||
- A rename onto an occupied slug answers 409 and changes nothing.
|
||||
- A rename attempt on a self-hosted instance from the customer portal answers
|
||||
400, the same status and constant the member endpoints already answer with.
|
||||
- The audit log carries `instance.renamed` with both slugs.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- Slug aliases or redirects from the old host. The control plane resolves one
|
||||
slug per instance, and an alias table is a second identity to keep correct for
|
||||
the sake of stale bookmarks.
|
||||
- Renaming from inside the control plane's own `/settings`. HQ owns instance
|
||||
identity, the same way it owns licences and `hq`-sourced users; a second
|
||||
writer would need the same collision handling and the same cooldown.
|
||||
- Any change to the licence, subscription or Paddle line items.
|
||||
@@ -0,0 +1,297 @@
|
||||
# Windows agent parity: OS updates and workloads
|
||||
|
||||
Date: 2026-08-13
|
||||
|
||||
## Goal
|
||||
|
||||
Bring the Windows agent up to the Linux agent on two subsystems: OS update
|
||||
check/apply, and the workload registry (collection, control, logs). Everything
|
||||
else about the Windows agent stays as it is.
|
||||
|
||||
Out of scope, deliberately:
|
||||
|
||||
- **Package inventory and CVE findings.** `trivy-db` carries no Windows feed, so
|
||||
a Windows finding needs a different source, a different matcher and a
|
||||
different version comparison. That is its own project, and until it exists a
|
||||
Windows host correctly reports `status: unsupported` rather than "0 findings".
|
||||
- **SSH key management on Windows.** `administrators_authorized_keys` is a real
|
||||
possibility but a separate decision.
|
||||
- **winget.** Third-party app upgrades are a different question from OS
|
||||
patching, and winget is absent on Server Core and older builds.
|
||||
|
||||
## Current state
|
||||
|
||||
The Windows agent registers, heartbeats, reports inventory, runs workflow steps
|
||||
through PowerShell, relays console connections and self-updates via MSI. Four
|
||||
gates stop it doing more:
|
||||
|
||||
| Gate | Location |
|
||||
| --- | --- |
|
||||
| `runtime.GOOS != "linux"` early return | `agent/internal/workloads/workloads.go`, `agent/internal/sync/workloads.go` (twice) |
|
||||
| package-manager detection finds nothing | `agent/internal/updates/updates.go` (`detectPM`) |
|
||||
| hard error | `agent/internal/packages/packages.go` (out of scope here) |
|
||||
| `authorized_keys` write skipped | `agent/internal/sync/sync.go` (out of scope here) |
|
||||
|
||||
## Approach
|
||||
|
||||
The platform split moves into the agent, expressed as build tags following the
|
||||
existing `inventory/collect_linux.go` / `collect_windows.go` /
|
||||
`collect_other.go` precedent. The control plane stays OS-blind: `ReportWorkloads`,
|
||||
`ControlWorkload`, `WorkloadLogs` and `ApplyUpdates` need no changes at all,
|
||||
because a Windows service is reported as the same `unit` kind a systemd service
|
||||
is.
|
||||
|
||||
Build tags rather than `runtime.GOOS` switches so PowerShell command strings do
|
||||
not ship in the Linux binary, and so a platform left unimplemented is a compile
|
||||
error rather than a silent no-op at runtime.
|
||||
|
||||
## Updates
|
||||
|
||||
### Layout
|
||||
|
||||
```
|
||||
agent/internal/updates/
|
||||
updates.go # PackageUpdate; CheckAvailable/ApplyAll declared once
|
||||
updates_linux.go # existing detectPM, checkApt/DnfYum/Pacman/Zypper/Apk, ApplyAll
|
||||
updates_windows.go # Windows Update COM, driven through PowerShell
|
||||
updates_other.go # //go:build !linux && !windows — no-ops
|
||||
```
|
||||
|
||||
`updates_other.go` carries the build constraint for the same reason
|
||||
`inventory/collect_other.go` does: `_other` is not a GOOS suffix, so without the
|
||||
constraint the file compiles everywhere and collides.
|
||||
|
||||
### Checking
|
||||
|
||||
One PowerShell invocation, `-NoProfile -NonInteractive`, emitting JSON:
|
||||
|
||||
```powershell
|
||||
$searcher = (New-Object -ComObject Microsoft.Update.Session).CreateUpdateSearcher()
|
||||
$result = $searcher.Search("IsInstalled=0 and Type='Software' and IsHidden=0")
|
||||
```
|
||||
|
||||
The Windows Update COM API is used rather than the `PSWindowsUpdate` module
|
||||
because it is present on every supported Windows, needs no PowerShell Gallery
|
||||
install, and works unchanged against a WSUS server on an air-gapped fleet. The
|
||||
agent runs as `LocalSystem`, which has the rights the API requires.
|
||||
|
||||
Each result maps to a `PackageUpdate`:
|
||||
|
||||
| Field | Value |
|
||||
| --- | --- |
|
||||
| `Name` | the update Title |
|
||||
| `CurrentVersion` | empty |
|
||||
| `NewVersion` | the KB article ID, e.g. `KB5034123` |
|
||||
|
||||
`CurrentVersion` is empty because a Windows update is not a version bump of a
|
||||
named package, and inventing a current version would put a wrong string in front
|
||||
of an operator. The KB ID goes in `NewVersion` because it is the identifier
|
||||
people actually search for.
|
||||
|
||||
Timeout: 10 minutes. The first search after a boot contacts Microsoft Update and
|
||||
is routinely slow.
|
||||
|
||||
### Applying
|
||||
|
||||
The same COM session: `CreateUpdateDownloader` then `CreateUpdateInstaller`,
|
||||
over the updates returned by the search above, skipping any that require user
|
||||
input. EULAs are accepted programmatically; an update whose EULA cannot be
|
||||
accepted is skipped rather than failing the batch.
|
||||
|
||||
The operation fails when the installer's `ResultCode` is not 2 (succeeded) or 3
|
||||
(succeeded with errors).
|
||||
|
||||
Timeout: 60 minutes. A patch-Tuesday cumulative genuinely takes that long, and
|
||||
the Linux path's existing 5-minute cap is already tight.
|
||||
|
||||
**The agent never reboots the host.** A control plane silently restarting a
|
||||
production server is unrecoverable from the UI, and the reboot is a decision a
|
||||
person or a workflow makes. Instead the need for one is reported.
|
||||
|
||||
### Reboot required
|
||||
|
||||
A new field `reboot_required` on `InventoryReport`, added to
|
||||
`proto/vantage/v1/vantage.proto` and to both hand-written `pb` copies
|
||||
(`agent/internal/grpc/pb`, `server/internal/grpc/pb`) in the same commit.
|
||||
|
||||
It travels on the inventory report rather than the update report because it is a
|
||||
host property like the kernel version, and it is set on the **static** snapshot
|
||||
only — every 15 minutes rather than every 30 seconds. A host rebooted by hand
|
||||
clears the flag in a quarter of an hour instead of showing it for up to a full
|
||||
one, and the detection costs a PowerShell process on Windows, which is not
|
||||
something to spawn twice a minute forever.
|
||||
|
||||
It is set in `agentsync.runInventory`, not inside the `inventory` package, so
|
||||
`inventory` gains no dependency on `updates`.
|
||||
|
||||
Both platforms set it, since parity is free here:
|
||||
|
||||
- Linux: `/var/run/reboot-required` exists, or `dnf needs-restarting -r` exits
|
||||
non-zero.
|
||||
- Windows: the `Microsoft.Update.SystemInfo` COM object's `RebootRequired`
|
||||
property, falling back to the pending-reboot registry keys
|
||||
(`Component Based Servicing\RebootPending`,
|
||||
`WindowsUpdate\Auto Update\RebootRequired`,
|
||||
`Session Manager\PendingFileRenameOperations`).
|
||||
|
||||
`services.ReportInventory` stores it on `servers.inventory`.
|
||||
|
||||
## Workloads
|
||||
|
||||
### Layout
|
||||
|
||||
```
|
||||
agent/internal/workloads/
|
||||
workloads.go # Result, Collect, Hash — Collect calls collectUnits
|
||||
docker.go # unchanged, shared: shells to the docker binary
|
||||
systemd_linux.go # was systemd.go
|
||||
services_windows.go # new: Win32_Service collection
|
||||
control.go # shared validation; platform halves split out
|
||||
control_linux.go # docker/systemctl, /proc/self/cgroup own-container check
|
||||
control_windows.go # Start/Stop/Restart-Service, VantageAgent protection
|
||||
logs.go # shared: capLog, MaxLogLines, MaxLogBytes
|
||||
logs_linux.go # docker logs / journalctl
|
||||
logs_windows.go # docker logs / Get-WinEvent
|
||||
```
|
||||
|
||||
`Collect` loses its `runtime.GOOS != "linux"` return and calls
|
||||
`collectUnits(ctx)`, which is the systemd collector on Linux and the service
|
||||
collector on Windows. `runWorkloads` and `reportWorkloads` in
|
||||
`agent/internal/sync/workloads.go` lose all three of their platform returns.
|
||||
|
||||
`docker.go` stays shared and ungated. It shells to the `docker` binary, which
|
||||
behaves identically on Windows, so a Docker Desktop or Mirantis host reports its
|
||||
containers with no new code. `DockerOK` / `DockerError` keep their existing
|
||||
three-state meaning: not installed (the common case, not a fault), installed but
|
||||
not responding, and running nothing.
|
||||
|
||||
### Collecting Windows services
|
||||
|
||||
`Get-CimInstance Win32_Service` converted to JSON — not `Get-Service`, which
|
||||
exposes neither `PathName` nor `StartMode`, and the filter needs both.
|
||||
|
||||
A service is reported when its executable does **not** resolve under
|
||||
`%SystemRoot%\System32`, and it is running, failed, or has `StartMode=Auto`
|
||||
while stopped. This mirrors the systemd collector's intent: show what an
|
||||
operator installed, and show what is meant to be up but is not.
|
||||
|
||||
Path parsing strips surrounding quotes and trailing arguments before the
|
||||
`%SystemRoot%` comparison. `"C:\Program Files\X\x.exe" -service` is one path
|
||||
with one argument, and splitting naively on whitespace misfiles a substantial
|
||||
share of a real fleet.
|
||||
|
||||
Field mapping:
|
||||
|
||||
| Workload field | Source |
|
||||
| --- | --- |
|
||||
| `Kind` | `"unit"` |
|
||||
| `ID` | `Name` (the service name) |
|
||||
| `Name` | `DisplayName` |
|
||||
| `State` | `running` / `stopped` / `failed`, from `State` plus `ExitCode` |
|
||||
| `Health`, `Image`, `Stack`, `Ports`, `Restarts` | unset |
|
||||
|
||||
`Kind: "unit"` and the existing `systemd_ok` / `systemd_error` fields are reused
|
||||
rather than a `service` kind and `services_ok` fields being added. That would
|
||||
cost a proto change, both pb copies, the server model, the service layer and the
|
||||
web client, and would teach every existing consumer a second kind — to describe
|
||||
the same thing. The naming is corrected where it is read, in the UI, which knows
|
||||
the server's OS.
|
||||
|
||||
`State` values match the ones the UI already colours, so no web change is needed
|
||||
for the rows themselves.
|
||||
|
||||
### Protection
|
||||
|
||||
The protected set stays computed and enforced agent-side, as it is on Linux: the
|
||||
control plane may name a target, but the agent decides what it will do to
|
||||
itself.
|
||||
|
||||
On Windows the protected workload is the `VantageAgent` service — the NSSM
|
||||
service name written by `installer/setup.ps1` — matched case-insensitively,
|
||||
because Windows service names are. `detectOwnContainer` and its
|
||||
`/proc/self/cgroup` read move to `control_linux.go`; the Windows build returns
|
||||
no own-container ID.
|
||||
|
||||
`ErrProtected` still surfaces as HTTP 409 from the API, and the reported
|
||||
`Protected` flag remains a courtesy that greys the button rather than the
|
||||
boundary.
|
||||
|
||||
### Control
|
||||
|
||||
`Start-Service`, `Stop-Service -Force`, `Restart-Service -Force`, under the same
|
||||
90-second `controlTimeout`, with the error text taken from PowerShell's stderr.
|
||||
|
||||
`sc.exe` is avoided because it returns before the operation completes, which
|
||||
turns a timeout into a false success. `-Force` is required because
|
||||
`Stop-Service` without it refuses when other services depend on the target, and
|
||||
that refusal reads to an operator as a silent no-op.
|
||||
|
||||
### Logs
|
||||
|
||||
`Get-WinEvent` with a filter hashtable over the `System` and `Application` logs,
|
||||
provider names matching the service name, its display name, and
|
||||
`Service Control Manager`, newest first, capped by the requested tail.
|
||||
|
||||
Each event is formatted as `<ISO 8601 timestamp> <Level> <Message>`, which is
|
||||
the same shape `journalctl --output=short-iso` produces, so the log dialog needs
|
||||
no per-platform rendering.
|
||||
|
||||
Service Control Manager logs every service on the host under one provider, so
|
||||
its events are filtered client-side to those naming the target service.
|
||||
|
||||
`capLog` is shared and unchanged: 500 lines and 256KB, whichever binds first,
|
||||
trimmed from the front. There is still no follow mode.
|
||||
|
||||
An empty result returns an empty string and no error. A service that has logged
|
||||
nothing is normal, and an error there would read as a broken feature.
|
||||
|
||||
## Server and web
|
||||
|
||||
The server changes in one place: `services.ReportInventory` persists
|
||||
`reboot_required`.
|
||||
|
||||
The web changes in three, all keyed on the same `os_info` test
|
||||
`MaintenanceTab.tsx` already uses (`server.os_info?.toLowerCase().includes("windows")`)
|
||||
rather than on `os_type`. `os_type` is stored and serialised but unread by
|
||||
`web/` today, and introducing a second Windows test in the same component is how
|
||||
the two come to disagree. `WorkloadList` takes the result as a prop, since it
|
||||
receives only a `serverId`:
|
||||
|
||||
1. `web/components/workloads/WorkloadList.tsx` — takes an `isWindows` prop from
|
||||
the server detail page, and the systemd status lines become
|
||||
platform-worded. On Windows the error line reads "Windows services could not
|
||||
be read" and the "systemd is not in use on this server" line is not rendered
|
||||
at all. The empty-state line drops "on Linux only". The Docker lines are
|
||||
unchanged.
|
||||
2. Server detail — a `Reboot required` pill beside the update count when the
|
||||
flag is set, placed with the update panel because that is what caused it.
|
||||
3. The Updates panel's Windows copy describes a list of KB articles rather than
|
||||
package upgrades, since `current_version` is empty on that platform.
|
||||
|
||||
## Testing
|
||||
|
||||
The Windows collectors are, in substance, parsers of PowerShell output. Parsing
|
||||
is separated from invocation and table-tested against captured real output. The
|
||||
`agent` module has no tests at all today, so these are the first — they live
|
||||
beside the parsers as ordinary `_test.go` files, run with `go test ./...` from
|
||||
`agent/`, and need no new dependency:
|
||||
|
||||
- `Win32_Service` JSON, including a quoted path with arguments, a
|
||||
`%SystemRoot%\System32` service that must be filtered out, a stopped
|
||||
`StartMode=Auto` service that must be kept, and a failed service with a
|
||||
non-zero `ExitCode`.
|
||||
- Update searcher JSON, including an update with no KB ID.
|
||||
- `Get-WinEvent` JSON, including a Service Control Manager event for another
|
||||
service that must be filtered out.
|
||||
- Pending-reboot detection from registry key presence.
|
||||
|
||||
`capLog` and `Hash` are unchanged and gain no tests.
|
||||
|
||||
The invocation halves are verified by hand on a Windows host: check, apply,
|
||||
service start/stop/restart, a protected refusal on `VantageAgent`, and logs on
|
||||
both a chatty service and a silent one.
|
||||
|
||||
`GOOS=windows go build ./...` and `GOOS=linux go build ./...` both belong in the
|
||||
implementation plan as explicit steps — a build-tag split is exactly the change
|
||||
that compiles on the machine you are sitting at and nowhere else. CI already
|
||||
cross-builds the agent on release, so no workflow change is needed.
|
||||
@@ -0,0 +1,286 @@
|
||||
# Public status pages
|
||||
|
||||
Date: 2026-08-24
|
||||
|
||||
## Goal
|
||||
|
||||
Let an operator publish one or more public status pages from a Vantage
|
||||
instance, at `<slug>.vantage.<tld>/status/<page-id>`, showing the state of any
|
||||
monitors they choose, plus incidents and maintenance windows they author by
|
||||
hand. The pages are completely public: no session, no token, no login.
|
||||
|
||||
Out of scope, deliberately:
|
||||
|
||||
- **Custom domains** (`status.customer.com`). Needs certificate provisioning and
|
||||
a host-to-page lookup that bypasses `hostSlug` entirely. Its own sub-project.
|
||||
- **Per-page themes.** `web/` is locked dark by design and a public page is not
|
||||
the place to break that.
|
||||
- **Subscriber notifications.** Email or webhook on incident updates is a
|
||||
notification subsystem, and one already exists for monitors; wiring the two
|
||||
together is a separate decision.
|
||||
- **SLA reporting.** Uptime percentages are shown; contractual SLA calculation
|
||||
with credits and exclusions is a different product.
|
||||
|
||||
## Current state
|
||||
|
||||
Everything needed to draw a status page already exists and is already scoped by
|
||||
instance:
|
||||
|
||||
| Data | Where |
|
||||
| --- | --- |
|
||||
| Monitor identity and live state | `models.Monitor`, `Monitor.State` |
|
||||
| Outage records | `models.Incident`, opened when a monitor flips down |
|
||||
| Hourly uptime history | `models.Rollup` (`monitor_rollups`) |
|
||||
| Sub-hour history | `models.MonitorSample`, TTL-expired |
|
||||
| Instance from hostname | `auth.InstanceFromHost`, 60s cached |
|
||||
|
||||
Three things do not exist: any concept of a page, any operator-authored
|
||||
incident, and any unauthenticated read path. The third is the constraint that
|
||||
shapes the rest — every route under `/api` carries `auth.Middleware`,
|
||||
`RequireScopes`, `RateLimitTokens` and `RequireActiveLicense` by virtue of where
|
||||
it is mounted, and `AssertScopeMapComplete` fails boot on an `/api` route with
|
||||
no scope entry.
|
||||
|
||||
## Approach
|
||||
|
||||
Two new collections hold the page and the authored incidents. A single
|
||||
assembly function reads them alongside the existing monitor data and emits a
|
||||
purpose-built public struct. The public route is mounted outside `/api`, is
|
||||
cached in Redis, and is rate limited per client address.
|
||||
|
||||
The redaction boundary is the assembly function, and it is the security
|
||||
property of this whole feature.
|
||||
|
||||
## Data model
|
||||
|
||||
Both collections carry `instance_id` and both must be added to
|
||||
`services.ScopedCollections`, or their rows outlive a deleted instance.
|
||||
|
||||
### `status_pages`
|
||||
|
||||
One document per page. It is read whole, always, so its structure is embedded
|
||||
rather than joined: one page is one Mongo read is one cache fill.
|
||||
|
||||
```
|
||||
_id, instance_id
|
||||
page_id // operator-chosen slug, [a-z0-9-], 3-40 chars
|
||||
title, description, logo_url
|
||||
published bool
|
||||
banner { enabled, level, text }
|
||||
sections [ { name, entries: [ { monitor_id, display_name } ] } ]
|
||||
created_at, updated_at
|
||||
```
|
||||
|
||||
Unique index on `(instance_id, page_id)`. The slug is operator-chosen rather
|
||||
than random because it is a URL handed to customers and printed on support
|
||||
pages; a random identifier would be unguessable and unmemorable in equal
|
||||
measure.
|
||||
|
||||
`published` exists so a page can be composed before anyone sees it. An
|
||||
unpublished page answers the same 404 as a page that does not exist — a
|
||||
distinct 403 would confirm it exists.
|
||||
|
||||
Sections are page-local and unrelated to `Monitor.Group`, which is a display
|
||||
label on the authenticated monitors list. One monitor may appear under "API" on
|
||||
the customer page and "Edge" on the partner page, under two different display
|
||||
names. That is the point of the override: a monitor's internal name is often
|
||||
not a name you want published.
|
||||
|
||||
The banner is three fields on the page rather than a collection, because it is
|
||||
one string with no lifecycle.
|
||||
|
||||
### `status_incidents`
|
||||
|
||||
Manual incidents and maintenance windows share one shape, because they share a
|
||||
timeline, an impact and a set of affected components; splitting them into two
|
||||
collections would duplicate all three.
|
||||
|
||||
```
|
||||
_id, instance_id, incident_id
|
||||
page_ids []string // which pages show it
|
||||
kind "incident" | "maintenance"
|
||||
title
|
||||
impact // none | minor | major | critical
|
||||
affected_monitors []string // monitor_ids
|
||||
status // incident: investigating | identified | monitoring | resolved
|
||||
// maintenance: scheduled | in_progress | completed
|
||||
scheduled_start, scheduled_end // maintenance only
|
||||
updates [ { at, status, body, author } ]
|
||||
started_at, resolved_at, created_at, updated_at
|
||||
```
|
||||
|
||||
Updates are embedded for the same reason sections are: they are few, and they
|
||||
are never read apart from their incident.
|
||||
|
||||
`page_ids` is explicit rather than derived from `affected_monitors`. Deriving it
|
||||
would be less to fill in, but adding a monitor to a page later would
|
||||
retroactively republish old incidents to a new audience. An operator publishing
|
||||
to customers chooses that audience.
|
||||
|
||||
### Auto-incidents are derived, never copied
|
||||
|
||||
The existing `incidents` collection remains the only writer for
|
||||
monitor-detected outages. The public snapshot derives them at assembly time:
|
||||
filter to the monitors on the page, last 90 days, render as display name, start,
|
||||
end and duration.
|
||||
|
||||
`Incident.Cause` is dropped. It is where `dial tcp 10.0.0.5:5432: connect
|
||||
refused` lives.
|
||||
|
||||
Copying auto-incidents into `status_incidents` would be a second writer for the
|
||||
same fact, arriving by a different route with its own opportunity to disagree —
|
||||
the same argument that keeps `RefreshWorkloadsCmd` from returning workloads
|
||||
inline.
|
||||
|
||||
### Maintenance does not rewrite uptime
|
||||
|
||||
During a maintenance window, affected components render as "under maintenance"
|
||||
rather than down. The uptime percentage and the history bar still come from the
|
||||
rollups, unmodified.
|
||||
|
||||
Rollups are the durable record. Bending them so a page looks better is a lie
|
||||
pointed the other way, and the operator who later asks "what was our actual
|
||||
availability" gets an answer that was edited for publication.
|
||||
|
||||
## The redaction boundary
|
||||
|
||||
`services.BuildStatusSnapshot(instanceID, pageID)` is the only function that
|
||||
reads `monitors`, `incidents`, `monitor_rollups` and `status_incidents` on
|
||||
behalf of an anonymous caller, and it emits a purpose-built struct.
|
||||
|
||||
**`models.Monitor` is never marshalled to a public caller.** Target URL, host,
|
||||
port, method, keyword, `state.message`, `state.cert_expiry_at` and
|
||||
`channel_ids` all stay behind the boundary. A field added to `Monitor` next year
|
||||
is private by default rather than published by accident.
|
||||
|
||||
What the snapshot contains, per entry: display name, current status, uptime
|
||||
percentage over the last 90 days, and a 90-day history bar of one cell per day.
|
||||
A cell is up, down, under maintenance, or no-data — `no-data` for days before
|
||||
the monitor existed, which is a distinct thing from a day it was down. No
|
||||
latency, no addresses, no failure text.
|
||||
|
||||
## Public read path
|
||||
|
||||
```
|
||||
GET /public/status/:pageId
|
||||
```
|
||||
|
||||
Mounted on the gin root, not under `apiGroup`. Putting it under `/api` would
|
||||
require exempting it from authentication, scope enforcement, token rate
|
||||
limiting and the licence gate — four holes, each one something a later change
|
||||
can widen. Outside `/api` it needs none of them.
|
||||
|
||||
The instance is resolved from the request host through `auth.InstanceFromHost`.
|
||||
A host with no instance label, an unknown slug, an unknown page and an
|
||||
unpublished page all answer **404**, identically.
|
||||
|
||||
### The feature gate answers 200, not 403
|
||||
|
||||
Status pages are gated by a new `license.FeatureStatusPages = "status_pages"`,
|
||||
on both the authoring routes and the public read.
|
||||
|
||||
The public side checks inline rather than through `RequireFeature`, which
|
||||
aborts with a 403 JSON body. A public page needs to render an explanation:
|
||||
|
||||
```json
|
||||
{ "available": false, "reason": "feature_unavailable", "title": "Acme Status" }
|
||||
```
|
||||
|
||||
`reason` is `feature_unavailable` when the tier does not include the feature and
|
||||
`licence_inactive` when the licence has lapsed. The title is included so the
|
||||
page does not look broken; nothing else is.
|
||||
|
||||
**This is not only a server change.** The feature must be added to admin's
|
||||
`plans` rows per `(deployment, tier)`, or every instance reads it as absent and
|
||||
the feature ships dark.
|
||||
|
||||
### Cache
|
||||
|
||||
Redis key `vantage:status:<instance_id>:<page_id>` holds the assembled JSON with
|
||||
a 30-second TTL. N visitors cost one Mongo read regardless of traffic.
|
||||
|
||||
Authoring writes delete the key, so an operator posting an incident update sees
|
||||
it immediately rather than wondering for half a minute whether it saved.
|
||||
|
||||
Redis rather than Next ISR because with `replicaCount > 1` each `web` pod would
|
||||
cache separately and two visitors would see different states during an incident.
|
||||
|
||||
### Rate limit
|
||||
|
||||
Per client address, one-minute fixed window, 120 requests, 429 with
|
||||
`Retry-After` — the same shape as `RateLimitTokens`, including its most
|
||||
important property: **when Redis is unavailable, allow rather than deny.** A
|
||||
status page must survive the outage it exists to report.
|
||||
|
||||
### Trusted proxies
|
||||
|
||||
Nothing calls `r.SetTrustedProxies`, so gin trusts every proxy and
|
||||
`c.ClientIP()` takes `X-Forwarded-For` verbatim. That is spoofable per request,
|
||||
which makes a per-address limiter decorative.
|
||||
|
||||
This has not mattered so far because `ClientIP()` is only used for audit
|
||||
strings. It matters now, so this work adds a trusted-proxy configuration and
|
||||
sets it at boot. Without it the rate limit is theatre.
|
||||
|
||||
## Authoring API
|
||||
|
||||
Under `/api`, owner or admin, behind `RequireFeature("status_pages")`, every
|
||||
mutation audited:
|
||||
|
||||
```
|
||||
GET,POST /status-pages
|
||||
GET,PUT,DELETE /status-pages/:pageId
|
||||
GET,POST /status-pages/:pageId/incidents
|
||||
PUT,DELETE /status-pages/:pageId/incidents/:incidentId
|
||||
POST /status-pages/:pageId/incidents/:incidentId/updates
|
||||
```
|
||||
|
||||
This adds a ninth scope resource, `status:read` and `status:write`. The entries
|
||||
are required, not optional: `AssertScopeMapComplete` fails boot on an `/api`
|
||||
route with no scope entry, which is exactly the safeguard working.
|
||||
|
||||
Handlers need `@…` annotations and `openapi.json` must be regenerated and
|
||||
committed — `server-deploy.yml` runs `git diff --exit-code` against the
|
||||
committed copy, so a handler whose annotation drifted fails CI.
|
||||
|
||||
## Frontend
|
||||
|
||||
`web/app/status/[pageId]/page.tsx`, **outside the `(app)` route group**, so it
|
||||
inherits no sidebar, no session fetch and no auth redirect. Server-rendered
|
||||
against the Go endpoint, with a client refresh every 60 seconds.
|
||||
|
||||
`web/next.config.ts` gains a `/public/:path*` rewrite so that client refresh
|
||||
reaches the server.
|
||||
|
||||
The page stays dark, like the rest of `web/`, and carries no hex values — the
|
||||
existing token palette covers every state it needs.
|
||||
|
||||
Authoring UI at `/status-pages` inside `(app)`, in the **Instance** sidebar
|
||||
group. It is `adminOnly`, and since the whole group is, a member sees the group
|
||||
disappear entirely rather than a labelled section with nothing under it.
|
||||
|
||||
## Testing
|
||||
|
||||
The snapshot tests are the ones that matter, because they are the redaction
|
||||
boundary made executable:
|
||||
|
||||
- `BuildStatusSnapshot` output contains no target URL or host, no
|
||||
`state.message`, no `incident.cause`, no `channel_ids`, no latency.
|
||||
- A monitor on no page never appears in any page's snapshot.
|
||||
- An unpublished page and an unknown page both 404.
|
||||
- Feature absent and licence inactive both return 200 with `available: false`
|
||||
and the matching `reason`.
|
||||
- A cache hit performs no Mongo read; an authoring write invalidates the key.
|
||||
- Slug validation: character set, length, uniqueness within an instance.
|
||||
- Maintenance window renders the component as under maintenance while leaving
|
||||
the uptime percentage untouched.
|
||||
|
||||
## Migration and rollout
|
||||
|
||||
No migration is needed — both collections are new and absent means empty. Index
|
||||
builders follow the `EnsureWorkflowIndexes` precedent and warn rather than being
|
||||
fatal: a missing index on a small collection degrades to a scan, which is no
|
||||
reason to refuse to serve the fleet.
|
||||
|
||||
The feature ships dark until the `status_pages` feature is added to the plan
|
||||
rows in admin.
|
||||
@@ -0,0 +1,330 @@
|
||||
# Control plane backup and restore
|
||||
|
||||
Date: 2026-09-07
|
||||
Status: approved, ready for implementation planning
|
||||
|
||||
## Problem
|
||||
|
||||
Vantage has no backup story. A self-hosted deployment holds its entire state in
|
||||
MongoDB and encrypts the sensitive half of it — SSH private keys, key
|
||||
passphrases, vault secrets, OIDC client secrets, RDP and VNC credentials — with
|
||||
AES-256-GCM under a single 32-byte key supplied as the `KEY_ENCRYPTION_KEY`
|
||||
environment variable.
|
||||
|
||||
That key is a bare value. It carries no identifier, is not wrapped, and is not
|
||||
recorded anywhere alongside the data it protects. Restoring a database without
|
||||
it produces a control plane whose every secret is permanently unreadable, and
|
||||
nothing in the product tells an operator this before it happens.
|
||||
|
||||
`mongodump` exists and operators can use it, but it says nothing about the
|
||||
encryption key, so the most common way to lose everything is to hold a perfectly
|
||||
good database dump and no key.
|
||||
|
||||
## Goal
|
||||
|
||||
A standalone command-line tool that backs up and restores a whole Vantage
|
||||
deployment, and that makes the key relationship impossible to get wrong by
|
||||
accident.
|
||||
|
||||
Explicitly not a goal: point-in-time recovery, incremental backups, built-in
|
||||
storage backends, encryption of the archive itself, per-tenant export, and
|
||||
backups scheduled from inside the server. Each is a separate decision and
|
||||
several are better served by tools the operator already has.
|
||||
|
||||
## Design
|
||||
|
||||
### Scope of a backup
|
||||
|
||||
One backup covers one MongoDB database: every collection in it, whether or not
|
||||
that collection is tenant-scoped. A deployment-level disaster recovery tool that
|
||||
skipped `migrations` or `vulndb_meta` would restore a database the server
|
||||
refuses to boot against.
|
||||
|
||||
Collections are enumerated live with `ListCollectionNames` rather than read from
|
||||
a hardcoded list. This is the opposite choice to `services.ScopedCollections`,
|
||||
and deliberately so: that list can afford to be hand-maintained because
|
||||
`AssertNoScopedCollectionMissed` fails boot when it drifts. A backup tool has no
|
||||
such assertion available, so a second hand-maintained registry would drift
|
||||
silently and the first symptom would be a restore missing a collection nobody
|
||||
noticed was added.
|
||||
|
||||
`--exclude` accepts collection names for the volume-heavy ones —
|
||||
`workflow_log_lines`, `monitor_samples`, `audit_logs`. Whatever is excluded is
|
||||
recorded in the manifest, so an archive can never claim to be complete when it
|
||||
is not.
|
||||
|
||||
Redis is not backed up. It holds sessions only; losing it logs everyone out and
|
||||
nothing else, which is already the documented behaviour. The restore output says
|
||||
so explicitly rather than leaving an operator to wonder.
|
||||
|
||||
### Where the code lives
|
||||
|
||||
Two units.
|
||||
|
||||
`shared/backup/` holds the logic: dump, restore, manifest construction, archive
|
||||
reading and writing, and key fingerprinting. It depends on the MongoDB driver
|
||||
and the standard library, and on no CLI framework. Keeping it in `shared/` and
|
||||
free of cobra is what lets `server` import it later if backups scheduled from
|
||||
inside the control plane are ever built, without pulling a command-line parser
|
||||
into the server binary.
|
||||
|
||||
`vantagectl/` is a new module in `go.work`, importing `shared`. It holds the
|
||||
cobra command tree and nothing else. A separate module rather than a package
|
||||
under `shared/` because adding cobra to `shared/go.mod` would put cobra and
|
||||
pflag into the module graph of `server`, `admin` and `sitesvc`, none of which
|
||||
use them. Binaries are unaffected — Go links only what is imported — but three
|
||||
`go.sum` files would grow and three CI builds would fetch a dependency they do
|
||||
not need. `agent/` is already a separate module for the same reason.
|
||||
|
||||
The tool imports nothing from `server/`. No `db.Col()`, no `services`, no config
|
||||
loader, and it never dials the REST or gRPC API. It needs only network reach to
|
||||
MongoDB, a database name, and `KEY_ENCRYPTION_KEY` in its own environment. This
|
||||
is what lets it run against a control plane that is down, half-migrated, or was
|
||||
deleted an hour ago — which is the only condition under which anyone runs a
|
||||
restore.
|
||||
|
||||
### Dump implementation
|
||||
|
||||
The dump is written against the MongoDB driver, not by shelling out to
|
||||
`mongodump`.
|
||||
|
||||
Two reasons. `server`'s runtime image is `scratch` and carries no shell and no
|
||||
mongo tools, so a wrapper would depend on a matching `mongodump` version being
|
||||
installed on whatever host runs the tool. And the manifest must be written by
|
||||
the same process that read the documents, or the fingerprint and per-collection
|
||||
checksums are claims about data the writer never saw.
|
||||
|
||||
The cost is that BSON round-tripping is ours to get right. Documents are written
|
||||
as raw BSON exactly as the driver returns them, without an intermediate map, so
|
||||
`ObjectId`, `Decimal128`, `DateTime`, binary subtypes and nulls survive
|
||||
unchanged. A round-trip test asserting byte-equal BSON is the guard.
|
||||
|
||||
### Archive format
|
||||
|
||||
A gzipped tar named `vantage-backup-<db>-<RFC3339>.tar.gz`:
|
||||
|
||||
```
|
||||
manifest.json
|
||||
collections/<name>.bson concatenated raw BSON documents
|
||||
indexes/<name>.json index specifications
|
||||
```
|
||||
|
||||
`manifest.json` carries:
|
||||
|
||||
| Field | Purpose |
|
||||
| --- | --- |
|
||||
| `format_version` | Currently `1`. Restore refuses an unknown version rather than guessing at it |
|
||||
| `created_at` | RFC3339, UTC |
|
||||
| `vantage_version` | Build stamp of the tool that wrote the archive |
|
||||
| `hostname` | Provenance; which machine produced this |
|
||||
| `mongo_db` | Source database name |
|
||||
| `mongo_server_version` | Restore warns on a major version gap |
|
||||
| `key_fingerprint` | `sha256` of the raw 32 key bytes, hex, or `null`. Never the key |
|
||||
| `collections[]` | Per collection: name, document count, uncompressed bytes, `sha256` of the `.bson` member |
|
||||
| `excluded[]` | Collection names passed to `--exclude` |
|
||||
|
||||
Per-collection checksums mean a truncated or corrupted archive is detected
|
||||
before a single document is written, rather than halfway through a restore.
|
||||
|
||||
### Key custody
|
||||
|
||||
The key never enters the archive. The archive is exactly as sensitive as a
|
||||
`mongodump` of the same database, and no more.
|
||||
|
||||
What the archive carries is `sha256` of the raw key bytes. A hash of the key
|
||||
proves identity without being a hint at the value, which is what allows an
|
||||
operator to answer "will this archive restore into this deployment" without
|
||||
holding both in front of them.
|
||||
|
||||
Backup refuses to run when `KEY_ENCRYPTION_KEY` is unset or malformed. An
|
||||
archive full of ciphertext whose key was never recorded is worse than no archive
|
||||
at all, because it looks like a backup. `--allow-no-key` exists for a deployment
|
||||
that genuinely stores no encrypted material; it stamps `key_fingerprint: null`,
|
||||
which restore then reports loudly rather than treating as a match.
|
||||
|
||||
Restore compares the archive's fingerprint against the key in the current
|
||||
environment:
|
||||
|
||||
- Fingerprints match: proceed.
|
||||
- Fingerprints differ: refuse, printing both.
|
||||
- Archive has a fingerprint, environment has no key: refuse.
|
||||
- `--ignore-key-mismatch`: proceed, having first printed exactly which
|
||||
collections hold ciphertext that will be undecryptable — `keys`, `secrets`,
|
||||
`auth_providers`, `console_sessions`, `settings`.
|
||||
|
||||
### Restore semantics
|
||||
|
||||
The order is fixed:
|
||||
|
||||
1. Read `manifest.json` and check `format_version`.
|
||||
2. Verify every archive member against its manifest checksum. Nothing is written
|
||||
before this passes.
|
||||
3. Apply the fingerprint rules above.
|
||||
4. Inspect the target: `ListCollectionNames` and document counts. A non-empty
|
||||
database is refused, printing what was found. `--force` proceeds.
|
||||
5. Per collection: under `--force`, drop it first; then bulk-insert in batches
|
||||
of 1000 with `ordered=false`.
|
||||
6. Replay index specifications from `indexes/<name>.json`, skipping `_id_`.
|
||||
7. Print a summary: collection, documents restored, indexes created.
|
||||
|
||||
Restore is not idempotent, and says so. A second run without `--force` is
|
||||
refused because step 4 now finds data. A restore interrupted during step 5
|
||||
leaves a partial database that the next run refuses to touch — correct, because
|
||||
the alternative is a silent merge. There are no merge or upsert semantics at
|
||||
all: merging two control planes reconciles nothing and produces a fleet that
|
||||
half works, and upserting by `_id` resurrects rows deleted since the backup,
|
||||
which for revoked keys and deleted users is a security regression wearing the
|
||||
costume of a convenience.
|
||||
|
||||
Index replay is fatal per collection when a unique index fails to build, and a
|
||||
warning when a non-unique one does. A unique index that cannot be created means
|
||||
the restored data violates it, and the unique indexes here — `(instance_id,
|
||||
email)`, instance slug, settings instance, the ESO token hash — are
|
||||
tenant-isolation properties rather than optimisations. The failure names the
|
||||
offending index.
|
||||
|
||||
### Destructive confirmation
|
||||
|
||||
Restore under `--force` requires a typed confirmation when stdin is a TTY.
|
||||
|
||||
When stdin is not a TTY — a Kubernetes Job, a CI step, a cron entry — the
|
||||
confirmation comes from `--confirm-db <name>`, whose value must equal the
|
||||
resolved target database name or restore refuses. Naming the database in the
|
||||
argument means a copy-pasted restore command carries its intended target with
|
||||
it and cannot destroy a different one.
|
||||
|
||||
A dynamic flag name containing the database name was considered and rejected:
|
||||
cobra registers flags before parsing, and the target database is not known at
|
||||
registration time.
|
||||
|
||||
### Command surface
|
||||
|
||||
```
|
||||
vantagectl root; prints help
|
||||
├── backup --out DIR|- --exclude a,b --allow-no-key
|
||||
├── restore ARCHIVE --force --confirm-db NAME --ignore-key-mismatch
|
||||
├── inspect ARCHIVE
|
||||
└── verify ARCHIVE
|
||||
```
|
||||
|
||||
Persistent flags on the root command, so every subcommand accepts them and they
|
||||
are documented once: `--mongo-uri` (env `MONGO_URI`) and `--db` (env `MONGO_DB`,
|
||||
falling back to the URI path). There is no `--log-level`: the tool's entire
|
||||
output is what it is telling the operator, and a level that could hide a key
|
||||
warning is worth not having.
|
||||
|
||||
Environment fallback is wired with an explicit `Changed` check on each flag
|
||||
rather than through viper. Viper is a configuration-file and remote-config
|
||||
system; this tool reads no configuration file, and pulling it in to call
|
||||
`os.Getenv` would make the largest dependency in the binary the one doing the
|
||||
smallest job.
|
||||
|
||||
`inspect` prints the manifest — when the archive was made, by what version,
|
||||
which collections it holds, how many documents, what was excluded, and the key
|
||||
fingerprint — and contacts no database. It is what an operator runs to find out
|
||||
whether an archive they have found is worth anything.
|
||||
|
||||
`verify` adds a live check: whether the archive's fingerprint matches the key in
|
||||
the current environment, and — when `--mongo-uri` is given — whether that key
|
||||
actually decrypts the target database. The second half is a probe: read one
|
||||
ciphertext field from `secrets`, `keys` or `auth_providers` and attempt to open
|
||||
it. A fingerprint comparison proves two archives agree; only a probe proves the
|
||||
key in hand opens the data in front of you. `verify` is the command that
|
||||
distinguishes "we have backups" from "we have backups that will restore", and
|
||||
the documentation recommends running it on a schedule.
|
||||
|
||||
The probe needs AES-256-GCM open, which today lives in
|
||||
`server/internal/services/crypto.go` and cannot be imported from another module.
|
||||
Rather than copy it — the exact hazard `CLAUDE.md` names around mirrored token
|
||||
blocks and `web/lib/targets.ts` — the primitives move to a new `shared/cryptobox`
|
||||
package, and `services/crypto.go` becomes a thin delegation that keeps its
|
||||
existing unexported function names and its `KEY_ENCRYPTION_KEY` lookup. One
|
||||
implementation of the cipher, two callers.
|
||||
|
||||
`--out -` streams the tarball to stdout, so piping into `aws s3 cp -`, `restic`
|
||||
or `age` covers storage and archive encryption without the tool growing backends
|
||||
of its own.
|
||||
|
||||
### Distribution
|
||||
|
||||
Three ways to run it, because the deployments that need it run Docker Compose,
|
||||
Kubernetes, or neither.
|
||||
|
||||
**Loose binary.** A new `.gitea/workflows/vantagectl-release.yml`, triggered on
|
||||
`vantagectl/v*` tags, shaped like `agent-release.yml`. Builds `linux/amd64`,
|
||||
`linux/arm64`, `darwin/arm64` and `windows/amd64` with `CGO_ENABLED=0`, writes
|
||||
`checksums.txt`, and creates a Gitea release.
|
||||
|
||||
**Container image.** `vantagectl/Dockerfile` — the repo's convention is a
|
||||
Dockerfile per module built from the repository root, because every Go module
|
||||
depends on `shared` through a replace directive — produces a `scratch`
|
||||
image holding the static binary and an explicitly copied `/tmp`, which the
|
||||
archive is staged in before compression — the same omission that silently
|
||||
disabled `vulnsched` on a scratch image. Pushed by `server-deploy.yml` as an
|
||||
eighth image.
|
||||
|
||||
```bash
|
||||
docker run --rm --network vantage_default \
|
||||
-e MONGO_URI -e MONGO_DB -e KEY_ENCRYPTION_KEY \
|
||||
-v /backups:/out \
|
||||
gitea.hostxtra.co.uk/mrhid6/vantagectl backup --out /out
|
||||
```
|
||||
|
||||
**Kubernetes.** The chart gains `backup.enabled`, defaulting to **false**,
|
||||
rendering a `CronJob` that runs the same image and mounts the existing MongoDB
|
||||
and `KEY_ENCRYPTION_KEY` secrets by reference rather than re-declaring them.
|
||||
Output goes to a PVC named in values. The default is off because a backup with
|
||||
nowhere durable to land is a false sense of safety and the chart cannot know
|
||||
where that is; `NOTES.txt` says so on install.
|
||||
|
||||
Restore in Kubernetes is the same image run as a one-shot `Job`. The chart ships
|
||||
no restore manifest: a restore is an operator decision with a confirmation
|
||||
attached to it, and must never be something a `helm upgrade` can trigger.
|
||||
|
||||
`server-deploy.yml`'s rebuild trigger table gains a `vantagectl` row —
|
||||
`vantagectl/`, `shared/`, `go.work` — which makes `shared/` fan out to four Go
|
||||
images rather than three. That table is already called out in `CLAUDE.md` as a
|
||||
place where a missed entry ships a stale image.
|
||||
|
||||
## Testing
|
||||
|
||||
`shared/backup` is tested against a real MongoDB, via `testcontainers-go` if the
|
||||
module graph tolerates it and otherwise behind a `MONGO_TEST_URI` environment
|
||||
variable that skips when unset.
|
||||
|
||||
Required cases:
|
||||
|
||||
- Round trip: seed one document of every awkward BSON type — `ObjectId`,
|
||||
`Decimal128`, `DateTime`, binary, null, nested arrays — back up, restore into
|
||||
a second database, assert byte-equal BSON.
|
||||
- A single corrupted byte in a `.bson` member causes restore to refuse before
|
||||
writing anything.
|
||||
- Fingerprint mismatch is refused; `--ignore-key-mismatch` proceeds and names
|
||||
the ciphertext-bearing collections.
|
||||
- A non-empty target is refused; `--force` replaces it.
|
||||
- An excluded collection is absent from the archive and named in the manifest.
|
||||
- A unique index that cannot be built aborts the restore, naming the index.
|
||||
|
||||
Fingerprint computation is a pure function and is tested without a database.
|
||||
|
||||
## Documentation
|
||||
|
||||
`docsite/docs/operations/backup-and-restore.md`, covering:
|
||||
|
||||
- What `KEY_ENCRYPTION_KEY` is, that it is not in the backup, and that losing it
|
||||
is unrecoverable. This comes first on the page, not as a note at the bottom.
|
||||
- The three run modes above, each as a command that can be copied.
|
||||
- A restore drill: restore into a scratch database and run `verify`, because an
|
||||
untested backup is a hypothesis.
|
||||
- What is not covered: Redis sessions, the vulnerability database (re-pulled
|
||||
automatically), and agent state on managed servers — agents reconnect on their
|
||||
own and `servers.agent_token_hash` is in the backup, so no re-enrolment is
|
||||
needed.
|
||||
|
||||
`CLAUDE.md` gains a section describing the tool, since a new module, a new
|
||||
image, a new workflow and a new chart toggle are each something that drifts
|
||||
quietly.
|
||||
|
||||
## Open questions
|
||||
|
||||
None. Every decision above was settled during design.
|
||||
@@ -55,8 +55,11 @@ It writes the config to `%ProgramData%\vantage\config.yaml`, installs the agent
|
||||
as a Windows service and starts it.
|
||||
|
||||
:::info Windows servers do not get SSH key management
|
||||
Windows agents register, report inventory and run workflow steps. Managing
|
||||
`authorized_keys` is a Linux-only feature.
|
||||
Windows agents register, heartbeat, report inventory, run workflow steps,
|
||||
serve the browser console, check and apply OS updates, and report workloads
|
||||
(services and containers). Managing `authorized_keys` is a Linux-only
|
||||
feature, and so is package inventory and CVE scanning — the vulnerability
|
||||
feeds this project uses carry no Windows data.
|
||||
:::
|
||||
|
||||
## 3. Watch it come up
|
||||
|
||||
@@ -96,8 +96,16 @@ rather than run in a half-prepared state.
|
||||
|
||||
## 4. Put a proxy in front
|
||||
|
||||
Point your reverse proxy at `web` on port `3000` and terminate TLS there. The
|
||||
web app reaches the API internally, so there is no need to publish port `8080`.
|
||||
Terminate TLS at your reverse proxy and route **one hostname to two backends**:
|
||||
|
||||
| Path | Backend |
|
||||
| -------------------------------------------------------------------------------- | ------------- |
|
||||
| `/api`, `/auth`, `/public`, `/install`, `/install.ps1`, `/update`, `/update.ps1` | `server:8080` |
|
||||
| everything else | `web:3000` |
|
||||
|
||||
Both rules are required. The web app forwards nothing to the API, so a proxy
|
||||
that sends the whole hostname to `web:3000` serves the interface and answers
|
||||
`404` to every request it makes — starting with the login form.
|
||||
|
||||
Agents connect to port `9090`. Vantage does not terminate TLS itself, so put
|
||||
that port behind your proxy too, with a certificate valid for the name in
|
||||
|
||||
@@ -14,7 +14,7 @@ self-hosted; what differs is the term on offer, not what you get.
|
||||
|
||||
| | Free | Professional | Enterprise |
|
||||
| --------------------- | --------- | ------------ | --------------------- |
|
||||
| Servers (base) | 3 | 3 | 10 |
|
||||
| Servers (base) | 3 | 5 | 10 |
|
||||
| Monitors | 3 | unlimited | unlimited |
|
||||
| Secret groups | 1 | unlimited | unlimited |
|
||||
| Notification channels | 1 | unlimited | unlimited |
|
||||
@@ -28,13 +28,14 @@ entitlement.
|
||||
|
||||
## Features
|
||||
|
||||
Three features are enabled per instance rather than bundled into a tier:
|
||||
Four features are enabled per instance rather than bundled into a tier:
|
||||
|
||||
| Feature | What it enables |
|
||||
| --------- | -------------------------------------------------------------------- |
|
||||
| Browser console | The [browser console](../vantage/browser-console.md) |
|
||||
| Single sign-on | [Sign-in through your identity provider](../vantage/settings.md#single-sign-on) |
|
||||
| Vulnerability scanning| [Package vulnerability scanning](../vantage/vulnerabilities.md) |
|
||||
| Feature | What it enables |
|
||||
| ---------------------- | ------------------------------------------------------------------------------- |
|
||||
| Browser console | The [browser console](../vantage/browser-console.md) |
|
||||
| Single sign-on | [Sign-in through your identity provider](../vantage/settings.md#single-sign-on) |
|
||||
| Vulnerability scanning | [Package vulnerability scanning](../vantage/vulnerabilities.md) |
|
||||
| Status pages | [Public status pages](../vantage/status-pages.md) |
|
||||
|
||||
No tier includes them by default; you enable them on the instances that need
|
||||
them.
|
||||
|
||||
@@ -0,0 +1,233 @@
|
||||
---
|
||||
id: backup-and-restore
|
||||
title: Backup and restore
|
||||
sidebar_label: Backup and restore
|
||||
---
|
||||
|
||||
`vantagectl` is a separate command-line tool that backs up and restores the
|
||||
MongoDB database behind a Vantage control plane. It talks to MongoDB directly,
|
||||
never to the Vantage API, so it works against a control plane that is down,
|
||||
half-migrated, or gone — exactly the situation a backup tool has to survive.
|
||||
|
||||
For the store-level overview — what holds what, and why the database alone is
|
||||
not a backup — see [Backups](./backups.md). This page covers the tool.
|
||||
|
||||
:::danger The key comes first
|
||||
Vantage encrypts SSH private keys, key passphrases, vault secrets, SSO client
|
||||
secrets and console credentials with `KEY_ENCRYPTION_KEY`. **It is not in your
|
||||
backup, and it is not recoverable.** A database restored without it is
|
||||
permanently unreadable — not degraded, not partially readable, unreadable.
|
||||
|
||||
Store it wherever you store the credentials you could not rebuild: a password
|
||||
manager, a secrets vault outside this control plane, a piece of paper in a
|
||||
safe. Anywhere but next to the archive.
|
||||
:::
|
||||
|
||||
## What a backup holds
|
||||
|
||||
Every collection in the database, the index definitions each one needs to be
|
||||
useful again, and a SHA-256 **fingerprint** of `KEY_ENCRYPTION_KEY` — never the
|
||||
key itself. The fingerprint is what lets a later `restore` or `verify` tell you
|
||||
that the key you are holding is the wrong one, before it writes a database
|
||||
nobody can read.
|
||||
|
||||
## What it does not hold
|
||||
|
||||
- **Redis sessions.** Everyone signs in again after a restore, which is already
|
||||
true whenever Redis itself restarts.
|
||||
- **The vulnerability database.** It is re-pulled automatically on next boot.
|
||||
- **Agent state on managed servers.** Nothing needs re-enrolling: agents
|
||||
reconnect on their own, because `servers.agent_token_hash` — the thing an
|
||||
agent authenticates with — is itself in the backup.
|
||||
|
||||
:::note Pin the version
|
||||
The image is published on each `vantagectl/v*` release and tagged with that
|
||||
version; `:latest` also moves. Pin a version in anything scheduled. A restore
|
||||
is easier to reason about when you can say which build produced the archive and
|
||||
which one read it back.
|
||||
:::
|
||||
|
||||
## Taking a backup
|
||||
|
||||
The loose binary:
|
||||
|
||||
```bash
|
||||
export MONGO_URI=mongodb://localhost:27017
|
||||
export MONGO_DB=vantage
|
||||
export KEY_ENCRYPTION_KEY=<your 64-char hex key>
|
||||
vantagectl backup --out /backups
|
||||
```
|
||||
|
||||
The container:
|
||||
|
||||
```bash
|
||||
docker run --rm \
|
||||
-e MONGO_URI=mongodb://mongo:27017 \
|
||||
-e MONGO_DB=vantage \
|
||||
-e KEY_ENCRYPTION_KEY=<your 64-char hex key> \
|
||||
-v /backups:/backups \
|
||||
gitea.hostxtra.co.uk/mrhid6/vantage/vantagectl:0.1.0 backup --out /backups
|
||||
```
|
||||
|
||||
Kubernetes, as a scheduled `CronJob` the Helm chart can render for you:
|
||||
|
||||
```yaml
|
||||
backup:
|
||||
enabled: true
|
||||
schedule: "0 2 * * *"
|
||||
image: "gitea.hostxtra.co.uk/mrhid6/vantage/vantagectl:0.1.0"
|
||||
pvcName: "vantage-backups"
|
||||
```
|
||||
|
||||
`backup.enabled` defaults to `false`, and the chart refuses to render if it is
|
||||
turned on without both `backup.image` and `backup.pvcName` — a backup needs a
|
||||
known image and somewhere durable to land, and guessing at either is worse than
|
||||
refusing to start. `backup.exclude` names collections to leave out (recorded in
|
||||
the archive's manifest, so an archive never claims to be complete when it is
|
||||
not), and `backup.successfulJobsHistoryLimit` / `backup.failedJobsHistoryLimit`
|
||||
/ `backup.resources` behave exactly as they do on any other `CronJob`.
|
||||
|
||||
`backup` refuses to run without `KEY_ENCRYPTION_KEY` set in the environment,
|
||||
unless you pass `--allow-no-key` — for a deployment that genuinely stores no
|
||||
encrypted data. Everywhere else, treat the refusal as the tool doing its job.
|
||||
|
||||
## Where to put the archive
|
||||
|
||||
`--out -` streams the tarball to stdout instead of writing a file, and every
|
||||
line of progress output goes to stderr — so piping the archive into something
|
||||
else is always safe, nothing progress-related lands in the stream.
|
||||
|
||||
Into `restic`:
|
||||
|
||||
```bash
|
||||
vantagectl backup --out - | restic backup --stdin --stdin-filename vantage.tar.gz
|
||||
```
|
||||
|
||||
Into S3:
|
||||
|
||||
```bash
|
||||
vantagectl backup --out - | aws s3 cp - s3://my-backups/vantage-$(date +%F).tar.gz
|
||||
```
|
||||
|
||||
An archive is as sensitive as a raw database dump — it carries every SSH key
|
||||
assignment, every secret group, every session-adjacent setting, in a form the
|
||||
right `KEY_ENCRYPTION_KEY` can decrypt. Whatever you pipe it into should
|
||||
encrypt it at rest; `vantagectl` itself does not.
|
||||
|
||||
## Checking a backup is real
|
||||
|
||||
```bash
|
||||
vantagectl verify /backups/vantage-backup-vantage-20260907T020000Z.tar.gz \
|
||||
--mongo-uri mongodb://localhost:27017 --db vantage
|
||||
```
|
||||
|
||||
Each line of output answers a different question:
|
||||
|
||||
- **`Archive`** — every member's checksum still matches; the tarball has not
|
||||
been truncated or corrupted.
|
||||
- **`Archive key`** / **`Your key`** — the fingerprint stored in the archive
|
||||
next to the fingerprint of the `KEY_ENCRYPTION_KEY` in your environment.
|
||||
- **`Key match`** — whether those two fingerprints agree.
|
||||
- **`Live probe`** — given `--mongo-uri`, `verify` goes one step further and
|
||||
decrypts a real ciphertext value from that database with the key you hold.
|
||||
A fingerprint match proves two archives agree about a key; only the probe
|
||||
proves the key in your hand actually reads the data.
|
||||
|
||||
`verify` exits non-zero the moment anything above is wrong, which is what makes
|
||||
it worth putting on a schedule — a backup job that "succeeded" last night is
|
||||
not the same claim as a backup that will actually restore.
|
||||
|
||||
## Looking inside an archive
|
||||
|
||||
`inspect` prints an archive's manifest and touches no database at all — no
|
||||
`--mongo-uri`, no key. It is what to run against an archive of unknown origin,
|
||||
before deciding whether it is the one you want:
|
||||
|
||||
```bash
|
||||
vantagectl inspect /backups/vantage-backup-vantage-20260907T020000Z.tar.gz
|
||||
```
|
||||
|
||||
It reports when the archive was taken and on which host, the Vantage and
|
||||
MongoDB versions behind it, the database it came from, the key fingerprint (or
|
||||
that it carries none), every collection with its document count and size, and
|
||||
anything `--exclude` left out. Opening the archive verifies every member's
|
||||
checksum on the way, so a corrupt archive fails here too.
|
||||
|
||||
Reach for `verify` instead when the question is whether the key you hold opens
|
||||
it; reach for `inspect` when the question is what it is.
|
||||
|
||||
## Restoring
|
||||
|
||||
`restore` expects the target database to be empty. Pointed at one that already
|
||||
holds data, it refuses outright: there are no merge semantics, because merging
|
||||
two control planes reconciles nothing and upserting old data over new would
|
||||
resurrect revoked keys and deleted users.
|
||||
|
||||
```bash
|
||||
vantagectl restore /backups/vantage-backup-vantage-20260907T020000Z.tar.gz \
|
||||
--mongo-uri mongodb://localhost:27017 --db vantage_restore
|
||||
```
|
||||
|
||||
To overwrite a database that is not empty, add `--force`, which drops each
|
||||
collection named in the archive before loading it. `--force` always needs a
|
||||
second assurance, in one of two forms:
|
||||
|
||||
- `--confirm-db NAME`, naming the target exactly. A mismatch is refused. This
|
||||
works everywhere — on a terminal and in a Kubernetes Job, a CI step or a cron
|
||||
entry alike — and is the form to script.
|
||||
- Nothing, on a terminal: `--force` alone prompts you to type the target
|
||||
database's name back, a deliberate pause before something destructive.
|
||||
|
||||
Without a terminal and without `--confirm-db`, `--force` is refused: there is
|
||||
nobody there to prompt. Naming the database in the command itself means a
|
||||
copy-pasted invocation carries its intended target with it and cannot destroy a
|
||||
different one by accident.
|
||||
|
||||
`--force` drops only the collections the archive carries. Anything else already
|
||||
in the target is left alone and named in a warning, so an archive taken with
|
||||
`--exclude workflow_log_lines` restored over a live database tells you the old
|
||||
log lines are still there, joined to freshly restored runs. Dropping them
|
||||
instead would delete data you never asked to delete.
|
||||
|
||||
`restore` also refuses when the archive's key fingerprint does not match the
|
||||
`KEY_ENCRYPTION_KEY` in your environment — see "When the key is wrong" below.
|
||||
|
||||
## The restore drill
|
||||
|
||||
An untested backup is a hypothesis, not a backup. Rehearse the whole path,
|
||||
monthly:
|
||||
|
||||
1. Restore last night's archive into a scratch database:
|
||||
```bash
|
||||
vantagectl restore /backups/vantage-backup-vantage-<date>.tar.gz \
|
||||
--mongo-uri mongodb://localhost:27017 --db vantage_drill
|
||||
```
|
||||
2. Run `verify` against the result to confirm the data that landed is actually
|
||||
readable with your current key:
|
||||
```bash
|
||||
vantagectl verify /backups/vantage-backup-vantage-<date>.tar.gz \
|
||||
--mongo-uri mongodb://localhost:27017 --db vantage_drill
|
||||
```
|
||||
3. Drop the scratch database. It served its purpose.
|
||||
|
||||
The failure this catches is not "the archive is corrupt" — `verify` alone
|
||||
catches that. It is "the archive is fine but nobody can actually stand a
|
||||
control plane back up from it," which only a real restore proves.
|
||||
|
||||
## When the key is wrong
|
||||
|
||||
If `restore` finds the archive's key fingerprint does not match the
|
||||
`KEY_ENCRYPTION_KEY` you are running with, it stops. Passing
|
||||
`--ignore-key-mismatch` proceeds anyway, but says plainly which collections
|
||||
will come back with ciphertext nobody can read:
|
||||
|
||||
- `keys` — SSH private keys and passphrases
|
||||
- `secrets` — the vault
|
||||
- `auth_providers` — OIDC/SSO client secrets
|
||||
- `console_sessions` — RDP/VNC credentials
|
||||
|
||||
There is no way to recover that ciphertext afterwards. If you have reached
|
||||
this point, the right key was lost along with the chance to read those rows —
|
||||
the fix is to re-enter each of them by hand (re-upload SSH keys, re-save vault
|
||||
secrets, reconfigure SSO), not to keep searching for a way to decrypt what is
|
||||
already in the database.
|
||||
@@ -24,7 +24,23 @@ values is permanently unreadable.
|
||||
Store the key somewhere other than the server it protects.
|
||||
:::
|
||||
|
||||
## Backing up MongoDB
|
||||
:::info Use `vantagectl`
|
||||
[**Backup and restore**](./backup-and-restore.md) is the supported way to take
|
||||
and restore a backup. It writes an archive that carries a fingerprint of
|
||||
`KEY_ENCRYPTION_KEY` — never the key — so a restore taken with the wrong key
|
||||
**refuses** rather than silently producing a database whose secrets nobody can
|
||||
read. It also checksums every archive member before writing anything, and
|
||||
refuses to restore into a database that already holds data. A plain
|
||||
`mongodump` does none of that: it records nothing about which key the data was
|
||||
encrypted under, so a restore from one succeeds even when the key is wrong and
|
||||
the failure only shows up later, as unreadable secrets.
|
||||
|
||||
The rest of this page, past the table above, describes the `mongodump` /
|
||||
`mongorestore` fallback for an operator who does not have `vantagectl`
|
||||
available. Prefer the linked page.
|
||||
:::
|
||||
|
||||
## Backing up MongoDB (fallback, without `vantagectl`)
|
||||
|
||||
With the bundled Mongo container:
|
||||
|
||||
@@ -33,6 +49,13 @@ docker compose exec -T mongo mongodump --archive --gzip --db vantage \
|
||||
> /backups/vantage-$(date +%F).archive.gz
|
||||
```
|
||||
|
||||
:::warning
|
||||
This archive records nothing about which `KEY_ENCRYPTION_KEY` it was taken
|
||||
under. Restoring it with the wrong key produces a database that looks intact
|
||||
and is not — every secret in it is silently unreadable until something tries
|
||||
to decrypt one.
|
||||
:::
|
||||
|
||||
Restoring:
|
||||
|
||||
```bash
|
||||
@@ -69,12 +92,13 @@ What it does **not** do is reconcile the world. After a restore:
|
||||
|
||||
| What | When |
|
||||
| ----------------- | ----------------------------------------------------- |
|
||||
| MongoDB dump | Nightly, retained per your policy |
|
||||
| Backup | Nightly, retained per your policy |
|
||||
| Environment file | On change, held in a password manager or secret store |
|
||||
| Restore rehearsal | Occasionally, into a throwaway host |
|
||||
|
||||
Rehearse a restore now and again. It is the step most often skipped, and the one
|
||||
that finds the problems.
|
||||
that finds the problems. See [Backup and restore](./backup-and-restore.md) for
|
||||
the drill, and for `verify`, which checks a backup is real without a restore.
|
||||
|
||||
## Cloud instances
|
||||
|
||||
|
||||
@@ -25,6 +25,7 @@ it is absent.
|
||||
| `VANTAGE_LICENSE` | no | | A licence supplied at startup, so an automated install does not have to paste one in |
|
||||
| `VANTAGE_TRIVY_DB_REF` | no | `ghcr.io/aquasecurity/trivy-db:2` | Where the vulnerability database is pulled from. Point it at a mirror for an air-gapped install |
|
||||
| `VANTAGE_VULNDB_DISABLED` | no | | `true` switches [vulnerability scanning](../vantage/vulnerabilities.md) off entirely. Findings already stored are still served, and still shown as stale |
|
||||
| `TRUSTED_PROXIES` | no | `10.0.0.0/8,172.16.0.0/12,192.168.0.0/16` | Comma-separated CIDRs or addresses of proxies allowed to set `X-Forwarded-For`. The shipped Docker Compose and Helm chart default to the private RFC1918 ranges, which covers Nginx Proxy Manager on the Docker bridge network and Traefik on a Kubernetes pod CIDR. An operator whose proxy sits on a public address must set this themselves, or every visitor behind it shares one address for rate-limiting purposes. Unset entirely (outside those shipped defaults) trusts none, so the client address is the direct peer. **On a LAN-only install, narrow this to your proxy's address.** The RFC1918 default trusts every private range, so a client on 192.168.0.0/16 reaching the server directly is itself a "trusted proxy" and can put whatever it likes in `X-Forwarded-For` — and, on the public status route, in `X-Forwarded-Host`. Behind a proxy on a public address, or with no proxy at all, that is not reachable; on a flat LAN it is |
|
||||
|
||||
:::danger `KEY_ENCRYPTION_KEY` has no recovery path
|
||||
It encrypts SSH private keys, vault secrets, OIDC client secrets and console
|
||||
|
||||
@@ -9,7 +9,7 @@ sidebar_label: Ports and networking
|
||||
| Port | Service | Who connects | Expose publicly |
|
||||
| ------- | ----------- | -------------------------------- | --------------- |
|
||||
| `3000` | web | Browsers, via your reverse proxy | Yes, behind TLS |
|
||||
| `8080` | server API | The web app | No, firewall it |
|
||||
| `8080` | server API | Your reverse proxy | Not directly — proxied |
|
||||
| `9090` | server gRPC | Agents | **Yes** |
|
||||
| `4822` | guacd | The server | No, firewall it |
|
||||
| `27017` | MongoDB | The server | No |
|
||||
@@ -20,8 +20,8 @@ sidebar_label: Ports and networking
|
||||
```mermaid
|
||||
flowchart LR
|
||||
B["Browser"] -->|HTTPS| P["Reverse proxy"]
|
||||
P --> W["web :3000"]
|
||||
W --> S["server :8080"]
|
||||
P -->|"everything else"| W["web :3000"]
|
||||
P -->|"/api /auth /public /install* /update*"| S["server :8080"]
|
||||
A["Agent on a managed server"] -->|"gRPC/TLS :9090, outbound"| S
|
||||
S --> G["guacd :4822"]
|
||||
G -->|"relayed over the :9090 stream"| A
|
||||
@@ -77,8 +77,20 @@ On a private network you can skip TLS instead, by setting `tls: false` in each
|
||||
|
||||
## Reverse proxy notes
|
||||
|
||||
- Point the proxy at `web:3000`. The web app reaches the API internally, so
|
||||
`8080` does not need publishing.
|
||||
- **The proxy routes two backends on one hostname**, and both are required:
|
||||
|
||||
| Path | Backend |
|
||||
| ------------------------------------------------------------- | ------------- |
|
||||
| `/api`, `/auth`, `/public`, `/install`, `/install.ps1`, `/update`, `/update.ps1` | `server:8080` |
|
||||
| everything else | `web:3000` |
|
||||
|
||||
The web app forwards nothing to the API. Sending the whole hostname to
|
||||
`web:3000` loads the interface and every request it makes answers `404` —
|
||||
including the login form.
|
||||
|
||||
- Both backends must be the **same** hostname and certificate. The browser
|
||||
calls `/api` relative to the page it is on, and the session cookie is
|
||||
host-only.
|
||||
- The console uses a **WebSocket** at `/api/console/tunnel`. A proxy that does
|
||||
not forward upgrade headers breaks the console and nothing else.
|
||||
- Workflow log streaming is a long-lived response. A short proxy read timeout
|
||||
|
||||
@@ -20,6 +20,13 @@ or is not 64 hex characters.
|
||||
|
||||
## Nobody can sign in
|
||||
|
||||
**Every request 404s and the interface loads fine.** Your reverse proxy sends
|
||||
the whole hostname to `web:3000`. `/api`, `/auth`, `/public`, `/install*` and
|
||||
`/update*` belong to `server:8080` and the web app forwards nothing — see
|
||||
[Ports and networking](./ports-and-networking.md#reverse-proxy-notes). The
|
||||
tell is `curl -si https://<your-host>/auth/bootstrap-status` returning HTML
|
||||
with `x-powered-by: Next.js` instead of JSON.
|
||||
|
||||
**`/setup` appears when users already exist.** The server is pointed at a
|
||||
different database than you think. Check the database name in `MONGO_URI`,
|
||||
which is taken from the end of the URI.
|
||||
@@ -95,6 +102,53 @@ instantaneous.
|
||||
- The keyword no longer appears in the response body.
|
||||
- Retries are `0`, so a single dropped packet flips the state.
|
||||
|
||||
### The check gets a 403, 429 or a CAPTCHA page
|
||||
|
||||
The endpoint is fine and answers a browser normally, but the monitor records a
|
||||
status it never sees by hand. Something between Vantage and the service is
|
||||
blocking automated traffic: a CDN, a WAF, a bot-protection product, a reverse
|
||||
proxy rule, or a rate limiter. The response usually comes from that layer and
|
||||
never reaches the origin at all, so nothing appears in the application's own
|
||||
logs.
|
||||
|
||||
Two things make it hard to spot. The check runs from the control plane's or the
|
||||
agent's address rather than yours, and those addresses are often datacenter
|
||||
ranges that bot protection scores badly. And a browser test proves nothing,
|
||||
because a browser is exactly what the blocking layer is willing to serve.
|
||||
|
||||
Every HTTP check Vantage makes identifies itself:
|
||||
|
||||
```
|
||||
User-Agent: Vantage-Monitor/1.0 (+https://vantage.hostxtra.co.uk)
|
||||
```
|
||||
|
||||
That string is the hook to allow the check through. In whichever product is
|
||||
doing the blocking, add a rule that skips bot protection, managed rules and rate
|
||||
limiting for requests carrying it — Cloudflare, AWS WAF, Azure Front Door,
|
||||
Akamai, Fastly, Imperva, Sucuri, ModSecurity, nginx and HAProxy all match on a
|
||||
request header. The shape of the rule is the same everywhere:
|
||||
|
||||
> If the host is *yours*, the path is *the one being monitored*, and the
|
||||
> User-Agent contains `Vantage-Monitor`, then skip the protection.
|
||||
|
||||
Three details are worth getting right:
|
||||
|
||||
- **Match on `contains`, not equality.** The version in the string moves. An
|
||||
exact match breaks silently on an upgrade, and the symptom is a monitor that
|
||||
goes down on deploy day.
|
||||
- **Keep the rule narrow.** Scope it to the specific host and path being
|
||||
monitored. A User-Agent is not a secret — anyone can send it — so a rule that
|
||||
skips protection site-wide on that string alone is a bypass you have
|
||||
published.
|
||||
- **Allow the source address too, where you can.** Combining the User-Agent with
|
||||
the checker's IP is stronger than either alone. Find the address in your
|
||||
blocking product's own event log; it is whichever client IP was blocked on the
|
||||
monitored path.
|
||||
|
||||
If the endpoint genuinely needs authentication rather than an exception, monitor
|
||||
a purpose-built health path that does not, and leave the protected paths
|
||||
protected.
|
||||
|
||||
## Notifications are not arriving
|
||||
|
||||
Use the channel **Test** button. It goes through the real delivery path, so a
|
||||
@@ -113,6 +167,32 @@ needs `host`, `port`, `from` and `to`, and Telegram needs both `token` and
|
||||
| Instance degraded despite a valid-looking licence | It expired more than a few days ago. Pasting a new one still works, which is how you recover |
|
||||
| Cannot enrol another server | The server allowance is reached. Raise it in HQ or remove one |
|
||||
|
||||
## A status page 404s or shows no data
|
||||
|
||||
**404, and it should be published.** Check the **Published** toggle on the
|
||||
page's editor — an unpublished page answers *not found* for everyone,
|
||||
including you, with no session exemption. Also check the host: the public URL
|
||||
is `<your-instance>.vantage.<yourdomain>/status/<page-id>`, the same
|
||||
per-instance subdomain everything else in Vantage uses. A wrong or missing
|
||||
subdomain resolves to no instance at all, which is also a 404.
|
||||
|
||||
Third possibility: `/public` is not routed to the server. Check with
|
||||
`curl -si https://<your-instance>.vantage.<yourdomain>/public/status/<page-id>`
|
||||
— JSON is correct, HTML carrying `x-powered-by: Next.js` means the proxy sent
|
||||
that prefix to the web app.
|
||||
|
||||
**Loads, but shows an explanation instead of components.** This is not a
|
||||
fault — it is the page working as designed. It means either the licence has
|
||||
lapsed (a self-hosted instance past its grace period, or a cloud instance
|
||||
between billing events) or the current tier does not include the **Status
|
||||
pages** feature. Fix the licence or the plan and the same link starts serving
|
||||
data again with no republish needed.
|
||||
|
||||
**One component reads `Unknown`.** The monitor behind it was deleted while
|
||||
still listed on the page. Nothing is checking it any more, so the page says so
|
||||
rather than showing a stale up or down. Remove the component from the page,
|
||||
or point it at a replacement monitor, in the page's editor.
|
||||
|
||||
## HQ portal problems
|
||||
|
||||
The portal is a hosted service, so problems with it are ours to fix rather than
|
||||
|
||||
@@ -64,6 +64,15 @@ Posts the alert as message content.
|
||||
|
||||
Port `465` uses implicit TLS; anything else uses STARTTLS.
|
||||
|
||||
### Credentials are never read back
|
||||
|
||||
The SMTP `password`, the Telegram `token` and the webhook, Slack and Discord
|
||||
`url`s come back from `GET /api/channels` as `••••••••` — a webhook URL is the
|
||||
authorisation to post to that channel, so it is treated as a credential like
|
||||
the rest. Writing that value back unchanged keeps the stored one, which is what
|
||||
lets you rename a channel without retyping its password. Anything else you send
|
||||
is written as given, so clearing the field clears the credential.
|
||||
|
||||
Alert emails look like the rest of the mail Vantage sends you.
|
||||
|
||||
## The message
|
||||
|
||||
@@ -89,10 +89,13 @@ metrics is normal rather than a fault.
|
||||
|
||||
### OS updates
|
||||
|
||||
Agents check for pending package updates hourly and report the count. From the
|
||||
server page you can:
|
||||
Agents check for pending package updates hourly and report the count — the
|
||||
machine's own package manager on Linux, the Windows Update COM API on Windows.
|
||||
From the server page you can:
|
||||
|
||||
- **Apply updates** runs the machine's own package manager and reports back.
|
||||
- **Apply updates** runs that check's install path and reports back. The agent
|
||||
never reboots the machine; if one is owed, a **reboot required** badge
|
||||
appears on the next inventory snapshot instead.
|
||||
- **Update agent** upgrades the Vantage agent on that machine. See
|
||||
[Agent updates](../operations/agent-updates.md).
|
||||
|
||||
@@ -107,8 +110,11 @@ Opens a browser SSH, RDP or VNC session. See [Browser console](./browser-console
|
||||
|
||||
## Windows servers
|
||||
|
||||
Windows agents register, run workflow steps and report inventory. They do not
|
||||
manage `authorized_keys`.
|
||||
Windows agents register, heartbeat, run workflow steps, report inventory,
|
||||
check and apply OS updates, report workloads (services and containers), and
|
||||
serve the browser console. They do not manage `authorized_keys`, and they are
|
||||
not covered by package inventory or CVE scanning — the vulnerability feeds
|
||||
this project uses carry no Windows data.
|
||||
|
||||
## Removing a server
|
||||
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
---
|
||||
id: status-pages
|
||||
title: Status pages
|
||||
sidebar_label: Status pages
|
||||
---
|
||||
|
||||
A status page is a public page reporting a chosen set of monitors as up-front
|
||||
components, with a 90-day history and an uptime percentage per component. It
|
||||
needs no session and no token to read — anyone with the link can open it,
|
||||
which is the point: it is what you hand a customer instead of an incident
|
||||
email.
|
||||
|
||||
Requires the **Status pages** licence feature. If the licence lapses, or the
|
||||
tier does not include the feature, the page keeps serving — it renders an
|
||||
explanation rather than data or a broken page, so a customer who follows an
|
||||
old link never sees an error.
|
||||
|
||||
## Creating a page
|
||||
|
||||
From **Status pages**, choose a page id and a title. The id is 3–40 characters
|
||||
of lowercase letters, digits and `-`, starting and ending with a letter or
|
||||
digit. It becomes part of the public URL:
|
||||
|
||||
```
|
||||
https://<your-vantage-address>/status/<page-id>
|
||||
```
|
||||
|
||||
On **Vantage Cloud** that address is your instance's own subdomain, so the page
|
||||
is at `https://<your-instance>.vantage.hostxtra.co.uk/status/<page-id>`.
|
||||
|
||||
On a **self-hosted** install it is whatever address you reach Vantage on —
|
||||
`https://vantage.acme.com/status/<page-id>`, or an IP and port on a LAN
|
||||
install. A self-hosted install serves exactly one Vantage instance, so no
|
||||
subdomain is needed to say which one you mean. The **Copy** control next to the
|
||||
page address in the editor gives you the exact URL for your install, which is
|
||||
the one to hand out.
|
||||
|
||||
**The page id cannot be changed after creation.** Once you have shared the
|
||||
link, changing the id would break it, so pick something you would still be
|
||||
happy with in a year — `platform`, `api`, a customer's own name for a
|
||||
dedicated page.
|
||||
|
||||
## Draft versus published
|
||||
|
||||
A new page starts unpublished. Unpublished pages answer *not found* to
|
||||
anyone who requests them, including you, from a browser without a session —
|
||||
so you can build out the components and copy before announcing it. Toggle
|
||||
**Published** when it is ready. Un-publishing later takes it back to *not
|
||||
found* rather than deleting anything.
|
||||
|
||||
**Delete page**, in the editor header, is the only way to correct a page id you
|
||||
regret — the id is fixed once created. It takes the page, its sections and its
|
||||
authored incidents with it; monitors and their history are untouched. If you
|
||||
only want the page off the internet, un-publish it instead.
|
||||
|
||||
## Sections and components
|
||||
|
||||
A page is organised into **sections** — arbitrary groupings such as "API" or
|
||||
"Region: EU" — each holding one or more **components**. A component is a
|
||||
monitor plus a **display name** you choose for this page.
|
||||
|
||||
The display name is never the monitor's own name unless you type it in. An
|
||||
internal monitor name ("prod-db-primary-eu1") is rarely what you want a
|
||||
customer reading; give it whatever name makes sense to them, and change it
|
||||
for a different page without touching the monitor.
|
||||
|
||||
If a monitor listed on a page is later deleted, its component still appears —
|
||||
reading `Unknown` rather than up or down, because nothing is checking it any
|
||||
more and claiming otherwise would be a false claim of health.
|
||||
|
||||
## What a visitor sees
|
||||
|
||||
- Component name, current state (up / down / under maintenance / pending /
|
||||
unknown) and a 90-day uptime percentage. **Pending** is a monitor that has
|
||||
been added but has not produced a result yet; **unknown** is one nothing is
|
||||
checking any more.
|
||||
- A 90-day history bar per component.
|
||||
- Any active incidents, upcoming maintenance, and a rolling history of both.
|
||||
- An optional banner across the top of the page, for anything you want said
|
||||
regardless of component state. It is one notice with one appearance — there
|
||||
are no severity levels to choose between.
|
||||
|
||||
A visitor never sees a target URL, host or port, the check's expected status
|
||||
or keyword, latency, a certificate expiry date, failure text, or which
|
||||
notification channel is attached. That is a deliberate boundary, not an
|
||||
oversight: nothing that would tell a stranger how your infrastructure is
|
||||
reachable is on this page.
|
||||
|
||||
## Incidents and maintenance
|
||||
|
||||
Two kinds of entries appear on a page's timeline:
|
||||
|
||||
- **Automatic** — a monitor going down opens an incident on any page that
|
||||
lists it, with no action from you. These appear the moment the monitor's
|
||||
state changes and close the moment it recovers.
|
||||
- **Authored** — an incident or maintenance window you create by hand, with
|
||||
its own title, impact and a set of affected components you choose. You
|
||||
post updates to it (Investigating → Identified → Monitoring → Resolved) as
|
||||
the situation develops, and each update is timestamped and kept on the
|
||||
page's history.
|
||||
|
||||
An authored incident is attached to one or more pages explicitly when you
|
||||
create it — it does not follow a monitor onto every page that monitor happens
|
||||
to be listed on.
|
||||
|
||||
### Scheduling maintenance
|
||||
|
||||
A maintenance window has a scheduled start and end (the end must be after the
|
||||
start) and moves through Scheduled → In progress → Completed. While a window
|
||||
is in progress and its affected components are within the scheduled time,
|
||||
those components are drawn as "under maintenance" instead of up or down.
|
||||
|
||||
**Maintenance changes how a day is drawn, never the uptime number itself.**
|
||||
The 90-day percentage is computed from what actually happened — a component
|
||||
that stayed up throughout a maintenance window still shows as up in its
|
||||
history, it is only the live status pill that reads "under maintenance" for
|
||||
the duration.
|
||||
|
||||
## Delay before an update appears
|
||||
|
||||
A visitor's read of a page is cached for up to 30 seconds, so posting an
|
||||
update or flipping Published does not necessarily change what a visitor sees
|
||||
instantly — though most authoring actions invalidate that cache immediately,
|
||||
so in practice it usually shows within a second or two. If a change genuinely
|
||||
does not appear, reloading after 30 seconds always will.
|
||||
@@ -4,23 +4,24 @@ title: Workloads
|
||||
sidebar_label: Workloads
|
||||
---
|
||||
|
||||
A **workload** is one Docker container or one systemd service. Each Linux server
|
||||
reports what it is running, and you can start, stop and restart those workloads,
|
||||
and read their recent logs, without opening a console.
|
||||
A **workload** is one Docker container or one service — a systemd unit on
|
||||
Linux, a Windows service on Windows. Every server reports what it is running,
|
||||
and you can start, stop and restart those workloads, and read their recent
|
||||
logs, without opening a console.
|
||||
|
||||
Available on every instance. No licence feature is required.
|
||||
|
||||
## What gets reported
|
||||
|
||||
Linux servers only, reported every 60 seconds.
|
||||
Every server, Linux and Windows, reported every 60 seconds.
|
||||
|
||||
- **Containers**: every container, running or not, with its image, published
|
||||
ports, health, restart count and the compose stack it belongs to.
|
||||
- **Services**: systemd units that are running, failed, or enabled but stopped.
|
||||
The operating system's own units are hidden, since a typical host has hundreds
|
||||
of them and they bury the ones you care about.
|
||||
|
||||
Windows servers report no workloads at all.
|
||||
ports, health, restart count and the compose stack it belongs to. Requires
|
||||
Docker (or Docker Desktop on Windows).
|
||||
- **Services**: systemd units on Linux that are running, failed, or enabled but
|
||||
stopped, and Windows services in the equivalent states. The operating
|
||||
system's own units and platform services are hidden, since a typical host
|
||||
has hundreds of them and they bury the ones you care about.
|
||||
|
||||
## Docker not in use is not an error
|
||||
|
||||
|
||||
+2
-1
@@ -29,6 +29,7 @@ const sidebars: SidebarsConfig = {
|
||||
"vantage/vulnerabilities",
|
||||
"vantage/workloads",
|
||||
"vantage/notification-channels",
|
||||
"vantage/status-pages",
|
||||
"vantage/secrets",
|
||||
"vantage/browser-console",
|
||||
"vantage/audit-log",
|
||||
@@ -48,7 +49,7 @@ const sidebars: SidebarsConfig = {
|
||||
{
|
||||
type: "category",
|
||||
label: "Operations",
|
||||
items: ["operations/upgrading", "operations/backups", "operations/agent-updates"],
|
||||
items: ["operations/upgrading", "operations/backups", "operations/backup-and-restore", "operations/agent-updates"],
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
@@ -168,6 +168,9 @@ message InventoryReport {
|
||||
uint64 swap_used = 7;
|
||||
repeated PartitionReport partitions = 8;
|
||||
string kernel = 9;
|
||||
// Set on static snapshots only. The agent never reboots; it reports that one
|
||||
// is owed and leaves the decision to a person or a workflow.
|
||||
bool reboot_required = 10;
|
||||
}
|
||||
|
||||
message InventoryReportResponse {
|
||||
|
||||
+22
-4
@@ -51,10 +51,10 @@ import (
|
||||
// @name Authorization
|
||||
// @description An API token, sent as "Bearer vt_…". Scoped and optionally expiring.
|
||||
|
||||
// @securityDefinitions.apikey esoAuth
|
||||
// @in header
|
||||
// @name Authorization
|
||||
// @description The External Secrets read token, rotated under Settings. It reaches /api/secrets/{group}/values and nothing else. It is a different credential from an API token, and the two must never be substituted for one another.
|
||||
// @securityDefinitions.apikey esoAuth
|
||||
// @in header
|
||||
// @name Authorization
|
||||
// @description The External Secrets read token, rotated under Settings. It reaches /api/secrets/{group}/values and nothing else. It is a different credential from an API token, and the two must never be substituted for one another.
|
||||
func main() {
|
||||
mongoURI := getEnv("MONGO_URI", "mongodb://localhost:27017")
|
||||
|
||||
@@ -162,6 +162,10 @@ func runSchemaSetup() {
|
||||
log.Printf("warning: failed to ensure workflow indexes: %v", err)
|
||||
}
|
||||
|
||||
if err := services.EnsureMonitorSampleIndexes(); err != nil {
|
||||
log.Printf("warning: failed to ensure monitor sample indexes: %v", err)
|
||||
}
|
||||
|
||||
if err := services.EnsureVulnIndexes(); err != nil {
|
||||
log.Printf("warning: failed to ensure vuln indexes: %v", err)
|
||||
}
|
||||
@@ -170,6 +174,10 @@ func runSchemaSetup() {
|
||||
log.Printf("warning: failed to ensure workload indexes: %v", err)
|
||||
}
|
||||
|
||||
if err := services.EnsureStatusPageIndexes(); err != nil {
|
||||
log.Printf("warning: failed to ensure status page indexes: %v", err)
|
||||
}
|
||||
|
||||
if err := services.EnsureAuditIndexes(); err != nil {
|
||||
log.Printf("warning: failed to ensure audit indexes: %v", err)
|
||||
}
|
||||
@@ -255,9 +263,19 @@ func serve() {
|
||||
})
|
||||
|
||||
r := gin.New()
|
||||
// Without this gin trusts every proxy and ClientIP() is whatever the
|
||||
// caller wrote in X-Forwarded-For. That was survivable while ClientIP()
|
||||
// only produced audit strings; the public status limiter makes it load
|
||||
// bearing. Empty means trust nobody, which is correct for a direct
|
||||
// exposure and wrong behind a proxy — hence the explicit setting.
|
||||
if err := r.SetTrustedProxies(api.TrustedProxies()); err != nil {
|
||||
log.Fatalf("trusted proxies: %v", err)
|
||||
}
|
||||
r.Use(gin.Recovery())
|
||||
r.Use(gin.LoggerWithConfig(gin.LoggerConfig{SkipPaths: []string{"/api/console/tunnel"}}))
|
||||
r.Use(corsMiddleware())
|
||||
services.SetStatusRedis(auth.Redis())
|
||||
|
||||
api.RegisterRoutes(r)
|
||||
|
||||
if err := api.AssertScopeMapComplete(r); err != nil {
|
||||
|
||||
@@ -34,7 +34,14 @@ func listChannels(c *gin.Context) {
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, channels)
|
||||
// Redacted here rather than in the service: the dispatchers read the same
|
||||
// documents and need the real credentials, so the masking belongs to the
|
||||
// boundary that hands them to a client.
|
||||
out := make([]models.NotificationChannel, 0, len(channels))
|
||||
for _, ch := range channels {
|
||||
out = append(out, ch.Redacted())
|
||||
}
|
||||
c.JSON(http.StatusOK, out)
|
||||
}
|
||||
|
||||
// createChannel godoc
|
||||
@@ -69,7 +76,7 @@ func createChannel(c *gin.Context) {
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusCreated, created)
|
||||
c.JSON(http.StatusCreated, created.Redacted())
|
||||
}
|
||||
|
||||
// updateChannel godoc
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -47,6 +47,10 @@ func RegisterRoutes(r *gin.Engine) {
|
||||
r.GET("/auth/oidc/:providerId/callback", auth.HandleSSOCallback)
|
||||
r.GET("/auth/providers", auth.HandleListPublicProviders)
|
||||
|
||||
// Completely public: no session, no token, no licence gate. Mounted here
|
||||
// rather than under /api precisely so that none of those apply.
|
||||
r.GET("/public/status/:pageId", RateLimitPublicStatus(), getPublicStatusPage)
|
||||
|
||||
apiGroup := r.Group("/api")
|
||||
apiGroup.Use(auth.Middleware())
|
||||
// Scope enforcement sits between authentication and the licence gate, and
|
||||
@@ -162,6 +166,8 @@ func RegisterRoutes(r *gin.Engine) {
|
||||
apiGroup.POST("/servers/:id/workloads/refresh", refreshServerWorkloads)
|
||||
apiGroup.POST("/servers/:id/workloads/:wid/action", auth.RequireRole("owner", "admin"), controlWorkload)
|
||||
apiGroup.GET("/servers/:id/workloads/:wid/logs", auth.RequireRole("owner", "admin"), getWorkloadLogs)
|
||||
|
||||
registerStatusPageRoutes(apiGroup)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@ package api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/auth"
|
||||
@@ -19,6 +20,7 @@ func registerMonitorRoutes(g *gin.RouterGroup) {
|
||||
g.DELETE("/monitors/:id", deleteMonitor)
|
||||
g.GET("/monitors/:id/incidents", getMonitorIncidents)
|
||||
g.GET("/monitors/:id/uptime", getMonitorUptime)
|
||||
g.GET("/monitors/:id/samples", getMonitorSamples)
|
||||
}
|
||||
|
||||
// listMonitors godoc
|
||||
@@ -111,7 +113,7 @@ func getMonitor(c *gin.Context) {
|
||||
// @Accept json
|
||||
// @Produce json
|
||||
// @Param id path string true "Monitor ID"
|
||||
// @Param body body object{name=string,type=string,target=models.MonitorTarget,interval_sec=int,runner=string,retries=int,enabled=bool,channel_ids=[]string} true "Fields to update"
|
||||
// @Param body body object{name=string,group=string,type=string,target=models.MonitorTarget,interval_sec=int,runner=string,retries=int,enabled=bool,channel_ids=[]string} true "Fields to update"
|
||||
// @Success 204
|
||||
// @Failure 400 {object} ErrorResponse
|
||||
// @Failure 500 {object} ErrorResponse
|
||||
@@ -121,6 +123,7 @@ func getMonitor(c *gin.Context) {
|
||||
func updateMonitor(c *gin.Context) {
|
||||
var body struct {
|
||||
Name *string `json:"name"`
|
||||
Group *string `json:"group"`
|
||||
Type *string `json:"type"`
|
||||
Target *models.MonitorTarget `json:"target"`
|
||||
IntervalSec *int `json:"interval_sec"`
|
||||
@@ -137,6 +140,9 @@ func updateMonitor(c *gin.Context) {
|
||||
if body.Name != nil {
|
||||
upd["name"] = *body.Name
|
||||
}
|
||||
if body.Group != nil {
|
||||
upd["group"] = *body.Group
|
||||
}
|
||||
if body.Type != nil {
|
||||
upd["type"] = *body.Type
|
||||
}
|
||||
@@ -217,6 +223,49 @@ func getMonitorIncidents(c *gin.Context) {
|
||||
c.JSON(http.StatusOK, incidents)
|
||||
}
|
||||
|
||||
// getMonitorSamples godoc
|
||||
//
|
||||
// @Summary Get a monitor's individual check results
|
||||
// @Description Raw check results for the last `minutes` minutes, oldest first. Samples expire after 48 hours; use the uptime rollups for longer ranges.
|
||||
// @Tags monitors
|
||||
// @Produce json
|
||||
// @Param id path string true "Monitor ID"
|
||||
// @Param minutes query int false "Window in minutes (default 60, max 2880)"
|
||||
// @Success 200 {array} models.MonitorSample
|
||||
// @Failure 404 {object} ErrorResponse
|
||||
// @Failure 500 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /monitors/{id}/samples [get]
|
||||
func getMonitorSamples(c *gin.Context) {
|
||||
m, err := services.GetMonitor(auth.InstanceID(c), c.Param("id"))
|
||||
if err != nil {
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
if m == nil {
|
||||
c.JSON(http.StatusNotFound, gin.H{"error": "monitor not found"})
|
||||
return
|
||||
}
|
||||
// Clamped rather than rejected: the window is a view setting, and the only
|
||||
// honest answer past the TTL is the shorter window anyway.
|
||||
minutes := 60
|
||||
if raw := c.Query("minutes"); raw != "" {
|
||||
if n, convErr := strconv.Atoi(raw); convErr == nil && n > 0 {
|
||||
minutes = n
|
||||
}
|
||||
}
|
||||
if max := int(services.MonitorSampleTTL.Minutes()); minutes > max {
|
||||
minutes = max
|
||||
}
|
||||
samples, err := services.MonitorSamples(auth.InstanceID(c), c.Param("id"), time.Now().Add(-time.Duration(minutes)*time.Minute))
|
||||
if err != nil {
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, samples)
|
||||
}
|
||||
|
||||
// getMonitorUptime godoc
|
||||
//
|
||||
// @Summary Get a monitor's uptime rollups
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/auth"
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/models"
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/services"
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/shared/license"
|
||||
"github.com/gin-gonic/gin"
|
||||
)
|
||||
|
||||
// publicStatusRateLimit is per client address per minute. Generous enough that
|
||||
// a busy page during an outage is unaffected, small enough that scanning for
|
||||
// page ids is not free.
|
||||
const publicStatusRateLimit = 120
|
||||
|
||||
// RateLimitPublicStatus counts requests per client address in a one-minute
|
||||
// fixed window, exactly as RateLimitTokens does — including the part that
|
||||
// matters most: when Redis is unavailable it allows rather than denies. A
|
||||
// status page must survive the outage it exists to report.
|
||||
func RateLimitPublicStatus() gin.HandlerFunc {
|
||||
return func(c *gin.Context) {
|
||||
rdb := auth.Redis()
|
||||
if rdb == nil {
|
||||
c.Next()
|
||||
return
|
||||
}
|
||||
window := time.Now().UTC().Unix() / 60
|
||||
key := "vantage:statusrl:" + c.ClientIP() + ":" + strconv.FormatInt(window, 10)
|
||||
|
||||
count, err := rdb.Incr(c.Request.Context(), key).Result()
|
||||
if err != nil {
|
||||
c.Next()
|
||||
return
|
||||
}
|
||||
if count == 1 {
|
||||
rdb.Expire(c.Request.Context(), key, 2*time.Minute)
|
||||
}
|
||||
if count > publicStatusRateLimit {
|
||||
c.Header("Retry-After", "60")
|
||||
c.AbortWithStatusJSON(http.StatusTooManyRequests, gin.H{
|
||||
"error": "too many requests",
|
||||
"code": "rate_limited",
|
||||
})
|
||||
return
|
||||
}
|
||||
c.Next()
|
||||
}
|
||||
}
|
||||
|
||||
// getPublicStatusPage is the only unauthenticated read of monitor data in the
|
||||
// product.
|
||||
//
|
||||
// It is mounted on the gin root rather than under /api on purpose: /api
|
||||
// carries auth.Middleware, RequireScopes, RateLimitTokens and
|
||||
// RequireActiveLicense by virtue of where it is mounted, and a public route
|
||||
// there would need four exemptions, each one a hole a later change can widen.
|
||||
//
|
||||
// Unknown host, unknown page and unpublished page all answer the same 404.
|
||||
//
|
||||
// It carries no @Router annotation deliberately. openapi.json declares a
|
||||
// single server of "/api", so a @Router of /public/status/{pageId} would be
|
||||
// published as /api/public/status/{pageId} — a path that does not exist, and
|
||||
// which would sit behind auth.Middleware if it did. The real address is:
|
||||
//
|
||||
// GET {scheme}://{instance-host}/public/status/{pageId}
|
||||
//
|
||||
// on the gin root, unauthenticated, rate limited per client address.
|
||||
//
|
||||
// @Summary Public status page
|
||||
// @Tags status
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Status page id"
|
||||
// @Success 200 {object} services.StatusSnapshot
|
||||
// @Failure 404 {object} ErrorResponse
|
||||
// @Failure 429 {object} ErrorResponse
|
||||
func getPublicStatusPage(c *gin.Context) {
|
||||
pageID := c.Param("pageId")
|
||||
inst, ok := publicStatusInstance(c)
|
||||
if !ok {
|
||||
// Every 404 on this route is indistinguishable to the caller by
|
||||
// design, so the log is the only place the three reasons are told
|
||||
// apart. It carries no monitor data and no page contents.
|
||||
log.Printf("public status: 404 page=%q reason=no_instance", pageID)
|
||||
c.JSON(http.StatusNotFound, gin.H{"error": "not found"})
|
||||
return
|
||||
}
|
||||
snap, err := services.PublicStatusSnapshot(inst.InstanceID, pageID)
|
||||
if errors.Is(err, services.ErrPageNotFound) {
|
||||
log.Printf("public status: 404 page=%q instance=%s reason=page_missing_or_unpublished", pageID, inst.InstanceID)
|
||||
c.JSON(http.StatusNotFound, gin.H{"error": "not found"})
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
log.Printf("public status: 500 page=%q instance=%s: %v", pageID, inst.InstanceID, err)
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": "internal error"})
|
||||
return
|
||||
}
|
||||
// Public and cacheable, but only briefly: an intermediary holding this for
|
||||
// minutes would show a resolved incident as ongoing.
|
||||
c.Header("Cache-Control", "public, max-age=30")
|
||||
c.JSON(http.StatusOK, snap)
|
||||
}
|
||||
|
||||
// publicStatusInstance resolves which instance a public request is for.
|
||||
//
|
||||
// The browser never reaches this handler directly: the request arrives from
|
||||
// the Next server, which forwards the visitor's host in X-Forwarded-Host
|
||||
// because the Host header cannot be set on a fetch (undici drops it silently,
|
||||
// as a forbidden header name). That makes X-Forwarded-Host a tenant selector,
|
||||
// so it is honoured only when the machine that opened the connection is one of
|
||||
// the configured trusted proxies.
|
||||
//
|
||||
// When the resulting host names no slug at all — vantage.acme.com,
|
||||
// status.acme.com, a bare IP — and the deployment is not cloud, the single
|
||||
// instance of that install is used. A self-hosted install has exactly one, and
|
||||
// without this every self-hosted status page 404s forever. More than one is a
|
||||
// refusal rather than a guess.
|
||||
func publicStatusInstance(c *gin.Context) (*models.Instance, bool) {
|
||||
// Host resolution is where this route fails silently: an untrusted peer
|
||||
// means X-Forwarded-Host is ignored and the request host is the Go
|
||||
// service's own name, which names no slug. Log the inputs and the branch
|
||||
// taken, so the 404 says which of the four it was.
|
||||
host := c.Request.Host
|
||||
xfh := firstForwarded(c.GetHeader("X-Forwarded-Host"))
|
||||
trusted := trustedPeer(c)
|
||||
if trusted && xfh != "" {
|
||||
host = xfh
|
||||
}
|
||||
log.Printf("public status: resolve peer=%s trusted=%t request_host=%q x_forwarded_host=%q using_host=%q slug=%q",
|
||||
c.RemoteIP(), trusted, c.Request.Host, xfh, host, auth.HostSlug(host))
|
||||
|
||||
if inst, ok := auth.InstanceForHost(host); ok {
|
||||
return inst, true
|
||||
}
|
||||
if slug := auth.HostSlug(host); slug != "" {
|
||||
// The host named an instance and that instance does not exist.
|
||||
log.Printf("public status: no instance for slug=%q (host=%q)", slug, host)
|
||||
return nil, false
|
||||
}
|
||||
if services.DeploymentMode() == license.DeploymentCloud {
|
||||
log.Printf("public status: host %q names no slug and deployment is cloud, refusing to guess", host)
|
||||
return nil, false
|
||||
}
|
||||
inst, ok := auth.SoleInstance()
|
||||
if !ok {
|
||||
log.Printf("public status: host %q names no slug and this deployment has no single instance", host)
|
||||
}
|
||||
return inst, ok
|
||||
}
|
||||
@@ -100,6 +100,7 @@ var routeScopes = map[string]string{
|
||||
"DELETE /api/monitors/:id": "monitors:write",
|
||||
"GET /api/monitors/:id/incidents": "monitors:read",
|
||||
"GET /api/monitors/:id/uptime": "monitors:read",
|
||||
"GET /api/monitors/:id/samples": "monitors:read",
|
||||
|
||||
// Channel routes, registered by registerChannelRoutes. Channels exist to
|
||||
// serve alerts, so they share the monitors scope rather than getting their
|
||||
@@ -155,6 +156,19 @@ var routeScopes = map[string]string{
|
||||
"GET /api/openapi.json": "settings:read",
|
||||
"GET /api/docs": "settings:read",
|
||||
"GET /api/docs/scalar.js": "settings:read",
|
||||
|
||||
// Status pages. Reading is status:read even though the pages themselves
|
||||
// are public, because these routes read the unpublished ones too.
|
||||
"GET /api/status-pages": "status:read",
|
||||
"POST /api/status-pages": "status:write",
|
||||
"GET /api/status-pages/:pageId": "status:read",
|
||||
"PUT /api/status-pages/:pageId": "status:write",
|
||||
"DELETE /api/status-pages/:pageId": "status:write",
|
||||
"GET /api/status-pages/:pageId/incidents": "status:read",
|
||||
"POST /api/status-pages/:pageId/incidents": "status:write",
|
||||
"PUT /api/status-pages/:pageId/incidents/:incidentId": "status:write",
|
||||
"DELETE /api/status-pages/:pageId/incidents/:incidentId": "status:write",
|
||||
"POST /api/status-pages/:pageId/incidents/:incidentId/updates": "status:write",
|
||||
}
|
||||
|
||||
// RequireScopes enforces routeScopes for token-authenticated requests and does
|
||||
|
||||
@@ -0,0 +1,312 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"net/http"
|
||||
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/auth"
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/models"
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/server/internal/services"
|
||||
"gitea.hostxtra.co.uk/mrhid6/vantage/shared/license"
|
||||
"github.com/gin-gonic/gin"
|
||||
)
|
||||
|
||||
func registerStatusPageRoutes(g *gin.RouterGroup) {
|
||||
// Owner or admin throughout: publishing a page is speaking to the public
|
||||
// in the instance's name. The feature gate sits alongside the role gate so
|
||||
// authoring and serving are gated by the same licence feature.
|
||||
sp := g.Group("/status-pages")
|
||||
sp.Use(auth.RequireRole("owner", "admin"), RequireFeature(license.FeatureStatusPages))
|
||||
|
||||
sp.GET("", listStatusPages)
|
||||
sp.POST("", createStatusPage)
|
||||
sp.GET("/:pageId", getStatusPage)
|
||||
sp.PUT("/:pageId", updateStatusPage)
|
||||
sp.DELETE("/:pageId", deleteStatusPage)
|
||||
sp.GET("/:pageId/incidents", listStatusIncidents)
|
||||
sp.POST("/:pageId/incidents", createStatusIncident)
|
||||
sp.PUT("/:pageId/incidents/:incidentId", updateStatusIncident)
|
||||
sp.DELETE("/:pageId/incidents/:incidentId", deleteStatusIncident)
|
||||
sp.POST("/:pageId/incidents/:incidentId/updates", appendStatusIncidentUpdate)
|
||||
}
|
||||
|
||||
// statusPageError maps the service errors onto codes once, so ten handlers do
|
||||
// not each invent their own. services.ErrPageInvalid covers every validation
|
||||
// failure in the status page and incident services — a missing title or an
|
||||
// invalid incident status is a 400, not a 500.
|
||||
func statusPageError(c *gin.Context, err error) {
|
||||
switch {
|
||||
case errors.Is(err, services.ErrPageNotFound), errors.Is(err, services.ErrIncidentNotFound):
|
||||
c.JSON(http.StatusNotFound, gin.H{"error": err.Error()})
|
||||
case errors.Is(err, services.ErrPageIDTaken):
|
||||
c.JSON(http.StatusConflict, gin.H{"error": err.Error()})
|
||||
case errors.Is(err, services.ErrInvalidPageID), errors.Is(err, services.ErrPageInvalid):
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
|
||||
default:
|
||||
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
|
||||
}
|
||||
}
|
||||
|
||||
// listStatusPages godoc
|
||||
//
|
||||
// @Summary List status pages
|
||||
// @Tags status-pages
|
||||
// @Produce json
|
||||
// @Success 200 {array} models.StatusPage
|
||||
// @Failure 500 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages [get]
|
||||
func listStatusPages(c *gin.Context) {
|
||||
pages, err := services.ListStatusPages(auth.InstanceID(c))
|
||||
if err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, pages)
|
||||
}
|
||||
|
||||
// createStatusPage godoc
|
||||
//
|
||||
// @Summary Create a status page
|
||||
// @Tags status-pages
|
||||
// @Accept json
|
||||
// @Produce json
|
||||
// @Param body body models.StatusPage true "Status page"
|
||||
// @Success 201 {object} models.StatusPage
|
||||
// @Failure 400 {object} ErrorResponse
|
||||
// @Failure 409 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages [post]
|
||||
func createStatusPage(c *gin.Context) {
|
||||
var p models.StatusPage
|
||||
if err := c.ShouldBindJSON(&p); err != nil {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
created, err := services.CreateStatusPage(auth.InstanceID(c), &p)
|
||||
if err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
services.LogEvent(auth.InstanceID(c), "status_page_created", actorFromCtx(c), "", "",
|
||||
"Status page '"+created.PageID+"' created")
|
||||
c.JSON(http.StatusCreated, created)
|
||||
}
|
||||
|
||||
// getStatusPage godoc
|
||||
//
|
||||
// @Summary Get a status page
|
||||
// @Tags status-pages
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Page id"
|
||||
// @Success 200 {object} models.StatusPage
|
||||
// @Failure 404 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages/{pageId} [get]
|
||||
func getStatusPage(c *gin.Context) {
|
||||
page, err := services.GetStatusPage(auth.InstanceID(c), c.Param("pageId"))
|
||||
if err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, page)
|
||||
}
|
||||
|
||||
// updateStatusPage godoc
|
||||
//
|
||||
// @Summary Update a status page
|
||||
// @Tags status-pages
|
||||
// @Accept json
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Page id"
|
||||
// @Param body body models.StatusPage true "Status page"
|
||||
// @Success 200 {object} models.StatusPage
|
||||
// @Failure 404 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages/{pageId} [put]
|
||||
func updateStatusPage(c *gin.Context) {
|
||||
var p models.StatusPage
|
||||
if err := c.ShouldBindJSON(&p); err != nil {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
updated, err := services.UpdateStatusPage(auth.InstanceID(c), c.Param("pageId"), &p)
|
||||
if err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
services.LogEvent(auth.InstanceID(c), "status_page_updated", actorFromCtx(c), "", "",
|
||||
"Status page '"+updated.PageID+"' updated")
|
||||
c.JSON(http.StatusOK, updated)
|
||||
}
|
||||
|
||||
// deleteStatusPage godoc
|
||||
//
|
||||
// @Summary Delete a status page
|
||||
// @Tags status-pages
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Page id"
|
||||
// @Success 204 "No Content"
|
||||
// @Failure 404 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages/{pageId} [delete]
|
||||
func deleteStatusPage(c *gin.Context) {
|
||||
if err := services.DeleteStatusPage(auth.InstanceID(c), c.Param("pageId")); err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
services.LogEvent(auth.InstanceID(c), "status_page_deleted", actorFromCtx(c), "", "",
|
||||
"Status page '"+c.Param("pageId")+"' deleted")
|
||||
c.Status(http.StatusNoContent)
|
||||
}
|
||||
|
||||
// listStatusIncidents godoc
|
||||
//
|
||||
// @Summary List authored incidents for a status page
|
||||
// @Tags status-pages
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Page id"
|
||||
// @Success 200 {array} models.StatusIncident
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages/{pageId}/incidents [get]
|
||||
func listStatusIncidents(c *gin.Context) {
|
||||
incs, err := services.ListStatusIncidents(auth.InstanceID(c), c.Param("pageId"))
|
||||
if err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
c.JSON(http.StatusOK, incs)
|
||||
}
|
||||
|
||||
// createStatusIncident godoc
|
||||
//
|
||||
// @Summary Create an incident or maintenance window
|
||||
// @Tags status-pages
|
||||
// @Accept json
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Page id"
|
||||
// @Param body body models.StatusIncident true "Incident"
|
||||
// @Success 201 {object} models.StatusIncident
|
||||
// @Failure 400 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages/{pageId}/incidents [post]
|
||||
func createStatusIncident(c *gin.Context) {
|
||||
var inc models.StatusIncident
|
||||
if err := c.ShouldBindJSON(&inc); err != nil {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
// The page in the path is always one of the pages the incident names, so
|
||||
// creating from a page cannot produce an incident that page never shows.
|
||||
if !contains(inc.PageIDs, c.Param("pageId")) {
|
||||
inc.PageIDs = append(inc.PageIDs, c.Param("pageId"))
|
||||
}
|
||||
created, err := services.CreateStatusIncident(auth.InstanceID(c), &inc)
|
||||
if err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
services.LogEvent(auth.InstanceID(c), "status_incident_created", actorFromCtx(c), "", "",
|
||||
"Status "+created.Kind+" '"+created.Title+"' created")
|
||||
c.JSON(http.StatusCreated, created)
|
||||
}
|
||||
|
||||
func contains(list []string, want string) bool {
|
||||
for _, v := range list {
|
||||
if v == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// updateStatusIncident godoc
|
||||
//
|
||||
// @Summary Update an incident or maintenance window
|
||||
// @Tags status-pages
|
||||
// @Accept json
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Page id"
|
||||
// @Param incidentId path string true "Incident id"
|
||||
// @Param body body models.StatusIncident true "Incident"
|
||||
// @Success 200 {object} models.StatusIncident
|
||||
// @Failure 404 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages/{pageId}/incidents/{incidentId} [put]
|
||||
func updateStatusIncident(c *gin.Context) {
|
||||
var inc models.StatusIncident
|
||||
if err := c.ShouldBindJSON(&inc); err != nil {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
updated, err := services.UpdateStatusIncident(auth.InstanceID(c), c.Param("incidentId"), &inc)
|
||||
if err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
services.LogEvent(auth.InstanceID(c), "status_incident_updated", actorFromCtx(c), "", "",
|
||||
"Status "+updated.Kind+" '"+updated.Title+"' updated")
|
||||
c.JSON(http.StatusOK, updated)
|
||||
}
|
||||
|
||||
// deleteStatusIncident godoc
|
||||
//
|
||||
// @Summary Delete an incident or maintenance window
|
||||
// @Tags status-pages
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Page id"
|
||||
// @Param incidentId path string true "Incident id"
|
||||
// @Success 204 "No Content"
|
||||
// @Failure 404 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages/{pageId}/incidents/{incidentId} [delete]
|
||||
func deleteStatusIncident(c *gin.Context) {
|
||||
if err := services.DeleteStatusIncident(auth.InstanceID(c), c.Param("incidentId")); err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
services.LogEvent(auth.InstanceID(c), "status_incident_deleted", actorFromCtx(c), "", "",
|
||||
"Status incident '"+c.Param("incidentId")+"' deleted")
|
||||
c.Status(http.StatusNoContent)
|
||||
}
|
||||
|
||||
// appendStatusIncidentUpdate godoc
|
||||
//
|
||||
// @Summary Post an update to an incident
|
||||
// @Tags status-pages
|
||||
// @Accept json
|
||||
// @Produce json
|
||||
// @Param pageId path string true "Page id"
|
||||
// @Param incidentId path string true "Incident id"
|
||||
// @Param body body StatusIncidentUpdateRequest true "Update"
|
||||
// @Success 200 {object} models.StatusIncident
|
||||
// @Failure 400 {object} ErrorResponse
|
||||
// @Failure 404 {object} ErrorResponse
|
||||
// @Security cookieAuth
|
||||
// @Security bearerAuth
|
||||
// @Router /status-pages/{pageId}/incidents/{incidentId}/updates [post]
|
||||
func appendStatusIncidentUpdate(c *gin.Context) {
|
||||
var body StatusIncidentUpdateRequest
|
||||
if err := c.ShouldBindJSON(&body); err != nil {
|
||||
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
|
||||
return
|
||||
}
|
||||
updated, err := services.AppendStatusIncidentUpdate(
|
||||
auth.InstanceID(c), c.Param("incidentId"), body.Status, body.Body, actorFromCtx(c))
|
||||
if err != nil {
|
||||
statusPageError(c, err)
|
||||
return
|
||||
}
|
||||
services.LogEvent(auth.InstanceID(c), "status_incident_update_posted", actorFromCtx(c), "", "",
|
||||
"Update posted to '"+updated.Title+"' ("+body.Status+")")
|
||||
c.JSON(http.StatusOK, updated)
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"net"
|
||||
"os"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"github.com/gin-gonic/gin"
|
||||
)
|
||||
|
||||
// TrustedProxies reads TRUSTED_PROXIES, a comma-separated list of CIDRs or
|
||||
// addresses. Unset means trust none: ClientIP() is then the peer address,
|
||||
// which is right for a direct exposure and means every request behind an
|
||||
// un-configured proxy shares one address for rate limiting. That is a visible
|
||||
// failure (one client limited) rather than an invisible one (no limit at all).
|
||||
//
|
||||
// This lives here rather than in main.go because the string has two consumers:
|
||||
// gin's own SetTrustedProxies, which main.go calls with it, and trustedPeer
|
||||
// below, which the public status page uses to decide whether to believe an
|
||||
// X-Forwarded-Host. One variable, one parser.
|
||||
func TrustedProxies() []string {
|
||||
v := strings.TrimSpace(os.Getenv("TRUSTED_PROXIES"))
|
||||
if v == "" {
|
||||
return nil
|
||||
}
|
||||
out := []string{}
|
||||
for _, p := range strings.Split(v, ",") {
|
||||
if p = strings.TrimSpace(p); p != "" {
|
||||
out = append(out, p)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
var (
|
||||
trustedNetsOnce sync.Once
|
||||
trustedNets []*net.IPNet
|
||||
)
|
||||
|
||||
func parsedTrustedNets() []*net.IPNet {
|
||||
trustedNetsOnce.Do(func() {
|
||||
for _, entry := range TrustedProxies() {
|
||||
if _, n, err := net.ParseCIDR(entry); err == nil {
|
||||
trustedNets = append(trustedNets, n)
|
||||
continue
|
||||
}
|
||||
// A bare address is a /32 or /128.
|
||||
if ip := net.ParseIP(entry); ip != nil {
|
||||
bits := 32
|
||||
if ip.To4() == nil {
|
||||
bits = 128
|
||||
}
|
||||
trustedNets = append(trustedNets, &net.IPNet{IP: ip, Mask: net.CIDRMask(bits, bits)})
|
||||
}
|
||||
}
|
||||
})
|
||||
return trustedNets
|
||||
}
|
||||
|
||||
// trustedPeer reports whether the immediate peer is one of the configured
|
||||
// proxies.
|
||||
//
|
||||
// It deliberately uses RemoteIP() rather than ClientIP(): ClientIP() is the
|
||||
// reconstructed *client* address, which is derived from the very headers this
|
||||
// function exists to decide whether to believe. X-Forwarded-Host selects a
|
||||
// tenant on the public status route, so it is only honoured when the machine
|
||||
// that actually opened the connection is trusted to have set it.
|
||||
func trustedPeer(c *gin.Context) bool {
|
||||
nets := parsedTrustedNets()
|
||||
if len(nets) == 0 {
|
||||
return false
|
||||
}
|
||||
ip := net.ParseIP(c.RemoteIP())
|
||||
if ip == nil {
|
||||
return false
|
||||
}
|
||||
for _, n := range nets {
|
||||
if n.Contains(ip) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user