diff --git a/CLAUDE.md b/CLAUDE.md index 192a6e0ca7..c7a565e8a6 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -25,6 +25,7 @@ Not everything is in this file. The following files contain critical details tha | Adding or modifying **HTTP endpoints** | `internal/server/server.go` — all routes are defined here — **plus** the handler's doc comments (why the route is gated the way it is lives there; copy the gate of the closest sibling only after reading it) and the integration's section in [docs/integrations.md](docs/integrations.md) | | Adding or modifying **CLI flags** | `cmd/explorer/main.go` — flag definitions and defaults | | Adding a **new CRD integration** (renderer, topology, discovery) | [docs/INTEGRATION_GUIDE.md](docs/INTEGRATION_GUIDE.md) — full checklist with collision gotchas | +| Working on the **CloudNativePG workspace** (`/cnpg`) | [docs/cnpg.md](docs/cnpg.md) — destinations, navigation (drawer trail, return label, `ctx` guard) and the certainty table: which source each fact comes from and what it reads when unknown. Data from `/api/cnpg/workspace` (per-kind coverage); derivations in `packages/k8s-ui/src/components/cnpg/workspace.ts` + `relations.ts`; screens in `web/src/components/cnpg/` | | Working on **local per-cluster integration settings** (Metrics, Argo CD, Cost in `~/.radar/clusters.json`) | [docs/configuration.md](docs/configuration.md#local-integration-connections) — store `internal/config/profiles.go`, resolve/update `internal/connections`, activation `internal/connectionruntime`, routes `GET/PUT /api/integrations/connections`. In local mode the older `PUT /api/integrations/{prometheus,argocd,cost}` return 409 | | Working on **GitOps** (Argo CD / Flux detail pages, operations, Terminating lifecycle, drift, per-resource health, remote destinations) | [docs/gitops.md](docs/gitops.md) — detail-page tabs, operation semantics, the Terminating severity ramp, nested navigation, single-cluster scope. Engine in `pkg/gitops/`, handlers `internal/server/gitops_handlers.go` | | Working on an **integration's reverse-lookup or actions** (Velero, CloudNativePG, Kyverno, Argo Rollouts, …) | That integration's section in [docs/integrations.md](docs/integrations.md) + the doc comments in `internal/server/_handlers.go` — both carry the per-integration gating and scope rules this file only summarizes | @@ -152,7 +153,7 @@ After `make -demo`, run `kubectl config use-context kind-radar--demo - RBAC reverse-lookup: `/api/rbac/subject/{kind}/{namespace}/{name}` (ServiceAccount, plus `usedByPods`) and `/api/rbac/subject/{kind}/{name}` (User/Group) — direct + group-inherited bindings and flattened effective rules; `/api/rbac/role/{kind}/{namespace}/{name}` (`_` for a ClusterRole's namespace) — the bindings that reference it; `/api/rbac/namespace/{namespace}` — backs the Namespace RBAC section (group-only ClusterRoleBindings deliberately excluded); `/api/rbac/whoami` — `SelfSubjectRulesReview` pass-through. All gate on `list rolebindings` AND `list clusterrolebindings`: **403 when either is denied, never a silent partial view** - Policy (Kyverno): `/api/policy/resource/{kind}/{ns}/{name}` (one resource's findings), `/api/policy/policies/{policy}` (every resource one policy recorded an outcome for). Report families are authorized **per subject scope** (`policyreports` cluster-wide ≠ `clusterpolicyreports`); findings from an unreadable family are dropped from lists AND counts, with the withheld count reported; `counts` describe the cluster while subject lists are capped and view-filtered. `/api/policy/policies/{policy}/queued` reads Kyverno's `UpdateRequest`s cluster-wide, gated on `list updaterequests` - Velero: `/api/velero/backupstoragelocations/{ns}/{name}/backups` (what a location holds; gated on `list backups`); `POST /api/velero/{backups|restores}/{ns}/{name}/messages` (a run's warnings/errors via a `DownloadRequest` — impersonated; needs a running Velero controller and object storage reachable from Radar, and reports which one failed) -- CloudNativePG: `/api/cnpg/imagecatalogs/{ns}/{name}/clusters`, `/api/cnpg/clusterimagecatalogs/{name}/clusters` (Clusters pinned to a catalog). Cluster-scoped catalogs are referenceable from any namespace, so that route reads cluster-wide gated on `list clusters` — a view-filtered answer would report "nothing uses this" before an edit +- CloudNativePG: `/api/cnpg/...` — the workspace, operator, catalog reverse-lookups, Cluster logs and activity. Routes, gates and coverage states are listed in [docs/cnpg.md](docs/cnpg.md#api); each is gated on the caller's own access, and a partial answer says what it withheld ## Key Patterns diff --git a/README.md b/README.md index 4ebe791242..e196a4cf18 100644 --- a/README.md +++ b/README.md @@ -548,7 +548,7 @@ Upgrade impact also gets list-only access to CSIStorageCapacities, FlowSchemas, | **Strimzi** | [KafkaConnector failure evidence](docs/integrations.md#strimzi-kafka-connectors) (connector/task status) | | **Velero** | Backup, Restore, Schedule, BackupStorageLocation, VolumeSnapshotLocation | | **External Secrets** | ExternalSecret, ClusterExternalSecret, SecretStore, ClusterSecretStore | -| **CloudNativePG** | Cluster, Backup, ScheduledBackup, Pooler | +| **CloudNativePG** | Cluster, Backup, ScheduledBackup, Pooler, Database, Publication, Subscription, ImageCatalog, ClusterImageCatalog, ObjectStore — plus a [workspace](docs/cnpg.md) for fleet, protection and declaration triage | | **Crossplane** | Managed Resources (any provider), Composite Resources, Claims, Provider, ProviderConfig, Function, Configuration, Composition, CompositionRevision, XRD | | **Kyverno** | Policy, ClusterPolicy, PolicyReport, ClusterPolicyReport | | **Sealed Secrets** | SealedSecret | diff --git a/docs/cnpg.md b/docs/cnpg.md new file mode 100644 index 0000000000..c2d43867f2 --- /dev/null +++ b/docs/cnpg.md @@ -0,0 +1,76 @@ +# CloudNativePG workspace + +A task-shaped view over [CloudNativePG](https://cloudnative-pg.io/) (CNPG): which PostgreSQL cluster needs attention, why, and what to inspect next — without assembling the story from ten separate CRD lists. The per-kind renderers, issue detection and audit check it builds on are described in [integrations.md](integrations.md#cloudnativepg). + +The workspace is read-only. It never writes to a cluster. + +## Where it lives + +CNPG stays inside **Resources**; there is no new global navigation item. When the `postgresql.cnpg.io` CRDs are discovered, the Resources sidebar's CloudNativePG group gains a **Workspace** block above its exact kinds: + +| Destination | Route | Job | Detail home for | +|---|---|---|---| +| Overview | `/cnpg` | The fleet: every Cluster with instances, replication, protection, declarations and its top problem. Defaults to **Needs attention**. | Cluster | +| Protection | `/cnpg/protection` | Recovery evidence per cluster, failed backups (7 days), destinations, schedules. | Backup, ScheduledBackup, ObjectStore | +| Declarations | `/cnpg/declarations` | Databases, Publications, Subscriptions and managed roles by cluster; declared vs reconciled. | Database, Publication, Subscription | +| Pooling | `/cnpg/pooling` | Poolers and the clusters they front. | Pooler | +| Operator | `/cnpg/operator` | Operator and plugin workloads, image catalogs, operator configuration. | ImageCatalog, ClusterImageCatalog | + +Destination badges count **affected clusters**, not findings, and follow the namespace filter (the sidebar says so). The exact kinds stay under a collapsible **Resource kinds** block, grouped by API group; on workspace screens it starts collapsed. + +Every CNPG kind's full detail is `/cnpg///` — reached from a row's **Open**, from the drawer's expand control, and by redirect from the generic `/workload/...` URL. The page keeps the workspace sidebar (its destination highlighted, the object nested under it) and uses Radar's detail view underneath: **Overview** is a composed summary, **Spec & status** is the kind's existing renderer, then YAML and the rest. A Cluster adds **Protection** (its recovery evidence), **Activity** (in place of Timeline) and merged instance **Logs**. + +## Navigation + +- **One drawer.** Rows inspect in the app's single drawer; `?drawer=kind:group:namespace:name` backs it, so refresh, share and Back restore it. Links inside the drawer append to that chain and show "← " at the top of the drawer. +- **Return vs location.** A full detail shows "← " only when it was reached by a drilldown (the label travels in history state); sidebar and global-nav hops are location changes and carry no return label. The crumb (`CloudNativePG / Protection / name`) always names the object's place, so a fresh tab has a parent without a fabricated previous task. +- **Context.** Detail URLs carry `ctx=` (added on first view when absent). After a context switch the page says " is not in " with **Switch back** and **Go to …** — Radar never opens a same-named object from another cluster. +- **Namespace filter** narrows collections and counts. An explicitly opened object stays open, with a note when it is outside the filter. + +## The certainty contract + +Every value is something the cluster reports, labelled with where it came from. When the cluster does not report something the UI says so; it never shows zero, "none" or green in its place. + +| Fact | Source | When it is not known | +|---|---|---| +| Instances, primary | `status.readyInstances`, `status.currentPrimary`, instance Pods (controller-owned by the Cluster's UID) | `–` | +| Replication | Pod readiness only | Always "lag unknown": readiness does not show whether a replica is streaming. Lag needs runtime data Radar does not read yet. | +| Schedule | ScheduledBackups targeting the Cluster (`spec.suspend` → suspended) | "No access to ScheduledBackups" when unreadable in that namespace | +| Destination | barman-cloud plugin `barmanObjectName`, in-tree `barmanObjectStore`, or volume snapshots | "No destination configured" | +| Last successful backup | Newest of: completed Backup CRs (7-day window plus the newest per cluster), ObjectStore `serverRecoveryWindow[...].lastSuccessfulBackupTime`, in-tree `status.lastSuccessfulBackup` (ignored for plugin clusters, where CNPG no longer sets it) — the winning source is shown | "None observed", or "No access to Backups" | +| WAL archiving | `ContinuousArchiving` condition | "Not reported" | +| Recovery window | Earliest point from ObjectStore `status.serverRecoveryWindow` for the cluster's server name. The latest point follows WAL archiving, not the last base backup, and no status reports it; it reads "not advancing" only while `ContinuousArchiving` is False | "Not reported" | +| Restore validation | A Cluster in the same namespace bootstrapped (`bootstrap.recovery`) from this cluster's store/server or one of its Backups, **with a ready instance** | "None recorded" (unknown tone) — Kubernetes records no restore tests, so this is never green. A matching cluster without a ready instance reads "Recovery declared in …". | +| ObjectStore upload health | **Inferred** from its user clusters' WAL archiving and recovery windows (ObjectStore has no status of its own) | "Unknown" | +| Declarations | `status.applied` (true / false / absent = pending); managed roles from `status.managedRolesStatus` (`reconciled`, `cannotReconcile`; anything else pending) | Pending, never failed | +| GitOps source | Argo CD / Flux labels and the Argo tracking annotation | "GitOps source not recorded" | +| Pooler pressure | — | "Not measured": needs PgBouncer metrics | +| ScheduledBackup cron | Shown verbatim | CNPG's cron is six-field (seconds first) and is never translated | + +Problems come from Radar's Issues engine (the same detections as `/issues`) plus the audit's `cnpgNoDeclarativeBackup`, worded "No declarative backup schedule" because that is all it proves. A cluster **needs attention** when it has an issue of warning or worse on itself, an instance Pod, or an object that references it. + +## Access + +All data comes from `GET /api/cnpg/workspace`, authorized **per kind**: namespaced kinds use a cluster-wide `list` or fall back per namespace; `ClusterImageCatalog` needs a cluster-scope `list`. Each kind reports coverage (`full`, `partial` with the namespaces read, `denied`, `syncing`, `error`, `notInstalled`). Issues and audit findings are withheld where the underlying kind is not covered — Pod evidence only reaches callers who can list Pods. Denied namespaces are named only when the caller supplied the namespace list. A partial or denied kind makes the screen show a coverage notice, and its facts read "No access" rather than none. + +`GET /api/cnpg/operator` reads operator and plugin Deployments (label `app.kubernetes.io/name=cloudnative-pg`, plugin Services labelled `cnpg.io/pluginName`) and the operator's config references. It ignores the namespace view filter (the operator lives in its own namespace), returns ConfigMap data only with `get configmaps`, and never reads Secrets. + +Cluster logs (`/api/cnpg/clusters/{ns}/{name}/logs`) need `get pods/log`; Activity (`.../activity`) drops events for kinds the caller cannot list. Deleted child objects stay attributed to their Cluster because Radar records the owning cluster on timeline events at ingestion; history recorded before that is marked incomplete. + +## Not in this version + +Runtime data (replication lag, sessions, locks, WAL and slots via the instance manager or Prometheus), Pooler pressure, and operations (Backup now, Switchover, Restart, Hibernate, Restore). + +## API + +Every route is gated on the caller's own access, and a partial answer names what it withheld rather than shrinking silently. + +- Workspace: `/api/cnpg/workspace` returns every CNPG kind (plus owner-validated instance Pods) with per-kind `coverage` (`full|partial|denied|notInstalled|syncing|error`, `partial` naming only in-scope denied namespaces), CNPG issues from the Issues engine and `cnpgNoDeclarativeBackup` audit findings, each withheld where the caller lacks coverage. Namespaced kinds follow the view filter and the capacity per-namespace `list` fallback; `ClusterImageCatalog` needs a cluster-scope `list`. Handler `internal/server/cnpg_workspace.go` +- Operator: `/api/cnpg/operator` returns the operator Deployments (`app.kubernetes.io/name=cloudnative-pg`) and plugin Deployments (served by Services labelled `cnpg.io/pluginName`) with image-tag version and readiness (`null` when unreported), plus the operator's ConfigMap/Secret/monitoring-queries references from its args and env. ConfigMap data only with `get configmaps`; the Secret is name-only, never read. Deployments and Services carry the workspace `coverage` states; deliberately ignores the namespace view filter (the operator lives in its own namespace). Handler `internal/server/cnpg_operator.go` +- Catalog reverse-lookup: `/api/cnpg/imagecatalogs/{ns}/{name}/clusters` and `/api/cnpg/clusterimagecatalogs/{name}/clusters` return the Clusters pinned to an image catalog, with the major each asks for and the image it actually resolved. Cluster-scoped catalogs are referenceable from any namespace, so the cluster-scoped route reads cluster-wide gated on `list clusters` — a view-filtered answer would report "nothing uses this" before an edit. +- Cluster logs: `/api/cnpg/clusters/{ns}/{name}/logs` (bounded snapshot) and `/logs/stream` (SSE, re-resolves instances every 5s) merge every instance Pod — label `cnpg.io/cluster` AND controller ownerRef to the Cluster's UID, never the label alone. Gated on `get clusters` + `list pods` + `get pods/log` before the Cluster lookup (404 after). `container` defaults to `postgres`, `tailLines` 200, `sinceTime` is converted to seconds and trimmed, `pod` must be a validated instance (400). Entries keep raw `content` and add `level`/`logger`/`message` parsed from the instance manager's JSON (`record.error_severity` wins over `level`). +- Cluster activity: `/api/cnpg/clusters/{ns}/{name}/activity?since=&limit=` reads the timeline store for the Cluster, its instance Pods (by owner) and CNPG children attributed by the retained `cnpg.io/cluster` label — `pkg/timeline.ExtractLabels` records it from the label or `spec.cluster.name` on CNPG-group objects and Pods, so deleted children stay attributed. K8s Event rows join by subject UID. Rows of a kind the caller can't `list` in the namespace are dropped; `oldest` is the namespace's retention floor and `attributionSince` the earliest labelled row — history before it cannot attribute deleted children. + +## Testing + +`make cnpg-demo` (read `scripts/cnpg-demo/README.md` first) produces WAL archiving failure, failed and unrecognised-phase Backups, failing declarations, a Pooler, both catalog kinds and an ObjectStore with a failing server — every state the workspace distinguishes, except successful restores. diff --git a/docs/integrations.md b/docs/integrations.md index 8de832bf33..5c5419a9a2 100644 --- a/docs/integrations.md +++ b/docs/integrations.md @@ -851,6 +851,8 @@ The source contract is Strimzi's [KafkaConnector status schema](https://strimzi. [CloudNativePG](https://cloudnative-pg.io/) (CNPG) is the Kubernetes operator for PostgreSQL, covering the full lifecycle from bootstrapping to monitoring, with high availability, automated failover, and backup management. +Beyond the per-kind views below, the CloudNativePG **workspace** (`/cnpg`) composes them into fleet, protection, declaration, pooling and operator screens — see [cnpg.md](cnpg.md). + ### What Radar Shows **Cluster Detail View:** diff --git a/internal/server/cnpg_cluster_activity.go b/internal/server/cnpg_cluster_activity.go new file mode 100644 index 0000000000..a3713a455f --- /dev/null +++ b/internal/server/cnpg_cluster_activity.go @@ -0,0 +1,251 @@ +package server + +import ( + "log" + "net/http" + "sort" + "strconv" + "strings" + "time" + + "github.com/go-chi/chi/v5" + + "github.com/skyhook-io/radar/internal/k8s" + "github.com/skyhook-io/radar/internal/timeline" + "github.com/skyhook-io/radar/pkg/resourceid" + pkgtimeline "github.com/skyhook-io/radar/pkg/timeline" +) + +const ( + cnpgActivityDefaultWindow = 24 * time.Hour + cnpgActivityDefaultLimit = 200 + cnpgActivityMaxLimit = 1000 + // cnpgActivityScanLimit bounds the rows read to attribute history. A + // namespace that outgrows it reports truncated rather than silently + // dropping its oldest attribution. + cnpgActivityScanLimit = 10000 +) + +// CNPGClusterActivityResponse is GET /api/cnpg/clusters/{namespace}/{name}/activity. +// Oldest is the earliest row the store still holds for the namespace — the +// floor below which absence means "not retained", not "didn't happen". +// AttributionSince is the earliest visible row that carries this Cluster's +// retained cnpg.io/cluster attribution; before it, deleted children cannot be +// attributed. Both are null when nothing is held. +type CNPGClusterActivityResponse struct { + Events []timeline.TimelineEvent `json:"events"` + Oldest *time.Time `json:"oldest"` + AttributionSince *time.Time `json:"attributionSince"` + Truncated bool `json:"truncated"` +} + +type cnpgActivityKind struct { + group, resource string +} + +// cnpgActivityKinds are the kinds whose rows can belong to one Cluster: the +// Cluster itself, its instance Pods, and every namespaced CNPG kind. +var cnpgActivityKinds = func() map[string]cnpgActivityKind { + out := map[string]cnpgActivityKind{"/Pod": {group: "", resource: "pods"}} + for _, k := range cnpgWorkspaceKinds { + if !k.clusterScoped { + out[k.group+"/"+k.kind] = cnpgActivityKind{group: k.group, resource: k.resource} + } + } + return out +}() + +func cnpgActivityKindNames() []string { + seen := map[string]bool{} + var out []string + for key := range cnpgActivityKinds { + _, kind, _ := strings.Cut(key, "/") + if !seen[kind] { + seen[kind] = true + out = append(out, kind) + } + } + sort.Strings(out) + return out +} + +// cnpgRowAttribution decides whether a timeline row is about the named +// Cluster. Rows about the Cluster match by identity; instance Pods by their +// controller owner; CNPG children by the retained cnpg.io/cluster label, which +// survives their deletion. liveUID is the UID of the Cluster that exists now +// under this name, or "" when none does; when set, only Pods it controlled +// count, so a previous same-named Cluster's instances don't merge into a +// recreated one's history. +func cnpgRowAttribution(e *timeline.TimelineEvent, name, liveUID string) (matched, labelled bool) { + group := resourceid.GroupFromAPIVersion(e.APIVersion) + if _, ok := cnpgActivityKinds[group+"/"+e.Kind]; !ok { + return false, false + } + labelled = e.Labels[pkgtimeline.CNPGClusterLabel] == name + switch { + case e.Kind == "Cluster" && group == cnpgGroup: + return e.Name == name, false + case e.Kind == "Pod" && group == "": + o := e.Owner + owned := o != nil && o.Kind == "Cluster" && o.Name == name && resourceid.GroupFromAPIVersion(o.APIVersion) == cnpgGroup && + (liveUID == "" || o.UID == liveUID) + return owned, owned && labelled + default: + return labelled, labelled + } +} + +// handleCNPGClusterActivity serves the Cluster's history from the timeline +// store: the Cluster, its instance Pods, and the CNPG objects attributed to it +// — including ones since deleted — with the K8s Events about each. Rows about +// a kind the caller cannot list in the namespace are dropped. +func (s *Server) handleCNPGClusterActivity(w http.ResponseWriter, r *http.Request) { + namespace, name := chi.URLParam(r, "namespace"), chi.URLParam(r, "name") + if !s.requireConnected(w) { + return + } + if noNamespaceAccess(s.getUserNamespaces(r, []string{namespace})) { + s.writeError(w, http.StatusForbidden, "no access to namespace "+namespace) + return + } + if !s.canRead(r, cnpgGroup, "clusters", namespace, "get") { + s.writeError(w, http.StatusForbidden, "no access to clusters.postgresql.cnpg.io in namespace "+namespace) + return + } + + now := time.Now() + since := now.Add(-cnpgActivityDefaultWindow) + if raw := r.URL.Query().Get("since"); raw != "" { + t, err := time.Parse(time.RFC3339, raw) + if err != nil { + s.writeError(w, http.StatusBadRequest, "invalid since "+strconv.Quote(raw)+" (expected RFC3339)") + return + } + since = t + } + limit := cnpgActivityDefaultLimit + if raw := r.URL.Query().Get("limit"); raw != "" { + n, err := strconv.Atoi(raw) + if err != nil || n <= 0 { + s.writeError(w, http.StatusBadRequest, "invalid limit "+strconv.Quote(raw)+" (expected a positive integer)") + return + } + limit = min(n, cnpgActivityMaxLimit) + } + + store := timeline.GetStore() + if store == nil { + s.writeError(w, http.StatusServiceUnavailable, "Timeline store not available") + return + } + clusterContext := k8s.ActiveClusterContext() + rows, err := store.Query(r.Context(), timeline.QueryOptions{ + Namespaces: []string{namespace}, + Kinds: cnpgActivityKindNames(), + APIGroups: []string{"", cnpgGroup, cnpgBarmanGroup}, + ClusterContext: clusterContext, + IncludeManaged: true, + IncludeK8sEvents: true, + Limit: cnpgActivityScanLimit, + }) + if err != nil { + log.Printf("[cnpg] Failed to query activity for %s/%s: %v", namespace, name, err) + s.writeError(w, http.StatusInternalServerError, err.Error()) + return + } + var liveUID string + if cache := k8s.GetResourceCache(); cache != nil { + if live, err := findCNPGCluster(r.Context(), cache, namespace, name); err == nil && live != nil { + liveUID = string(live.GetUID()) + } + } + scanCapped := len(rows) >= cnpgActivityScanLimit + + // Attribution is carried by the subject's own rows; K8s Event rows about + // a subject whose enrichment was already gone carry only its UID. + attributedUIDs := map[string]bool{} + matched := make([]bool, len(rows)) + labelled := make([]bool, len(rows)) + for i := range rows { + matched[i], labelled[i] = cnpgRowAttribution(&rows[i], name, liveUID) + if matched[i] && rows[i].UID != "" { + attributedUIDs[rows[i].UID] = true + } + } + + allowed := map[string]bool{} + var eventsAllowed *bool + canList := func(e *timeline.TimelineEvent) bool { + if e.Source == timeline.SourceK8sEvent { + if eventsAllowed == nil { + ok := s.canRead(r, "", "events", namespace, "list") + eventsAllowed = &ok + } + if !*eventsAllowed { + return false + } + } + key := resourceid.GroupFromAPIVersion(e.APIVersion) + "/" + e.Kind + ok, seen := allowed[key] + if !seen { + target, known := cnpgActivityKinds[key] + ok = known && s.canRead(r, target.group, target.resource, namespace, "list") + allowed[key] = ok + } + return ok + } + + resp := CNPGClusterActivityResponse{Events: []timeline.TimelineEvent{}} + seenIDs := map[string]bool{} + var windowed []timeline.TimelineEvent + for i := range rows { + e := &rows[i] + if !matched[i] && (e.UID == "" || !attributedUIDs[e.UID]) { + continue + } + if !canList(e) || seenIDs[e.ID] { + continue + } + seenIDs[e.ID] = true + if labelled[i] && (resp.AttributionSince == nil || e.Timestamp.Before(*resp.AttributionSince)) { + t := e.Timestamp.UTC() + resp.AttributionSince = &t + } + if e.Timestamp.Before(since) { + continue + } + windowed = append(windowed, *e) + } + sort.SliceStable(windowed, func(i, j int) bool { + if !windowed[i].Timestamp.Equal(windowed[j].Timestamp) { + return windowed[i].Timestamp.After(windowed[j].Timestamp) + } + return windowed[i].ID < windowed[j].ID + }) + resp.Truncated = scanCapped || len(windowed) > limit + if len(windowed) > limit { + windowed = windowed[:limit] + } + if windowed != nil { + resp.Events = windowed + } + + oldest, err := store.Query(r.Context(), timeline.QueryOptions{ + Namespaces: []string{namespace}, + ClusterContext: clusterContext, + IncludeManaged: true, + IncludeK8sEvents: true, + SequenceOrder: timeline.SequenceOrderAscending, + Limit: 1, + }) + if err != nil { + log.Printf("[cnpg] Failed to query retention floor for %s/%s: %v", namespace, name, err) + s.writeError(w, http.StatusInternalServerError, err.Error()) + return + } + if len(oldest) > 0 { + t := oldest[0].Timestamp.UTC() + resp.Oldest = &t + } + s.writeJSON(w, resp) +} diff --git a/internal/server/cnpg_cluster_history_test.go b/internal/server/cnpg_cluster_history_test.go new file mode 100644 index 0000000000..6c5f23aee6 --- /dev/null +++ b/internal/server/cnpg_cluster_history_test.go @@ -0,0 +1,525 @@ +package server + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/client-go/kubernetes" + "k8s.io/client-go/rest" + + "github.com/skyhook-io/radar/internal/auth" + "github.com/skyhook-io/radar/internal/k8s" + "github.com/skyhook-io/radar/internal/timeline" + pkgtimeline "github.com/skyhook-io/radar/pkg/timeline" +) + +func seedCNPGLogCluster(t *testing.T, ns string) { + t.Helper() + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + withUID(cnpgObj("postgresql.cnpg.io/v1", "Cluster", ns, "pg-orders", map[string]any{"instances": int64(2)}, nil), "orders-uid"), + cnpgObj("cluster.x-k8s.io/v1beta1", "Cluster", ns, "capi-only", nil, nil), + ) + owner := metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "orders-uid", Controller: boolPtr(true)} + stale := owner + stale.UID = "previous-incarnation" + replica := cnpgPod(ns, "pg-orders-2", "pg-orders", owner) + replica.Labels["cnpg.io/instanceRole"] = "replica" + seedCNPGPods(t, + cnpgPod(ns, "pg-orders-1", "pg-orders", owner), + replica, + cnpgPod(ns, "pg-orders-impostor", "pg-orders"), + cnpgPod(ns, "pg-orders-orphan", "pg-orders", stale), + ) +} + +func getCNPGLogs(t *testing.T, path string) (int, CNPGClusterLogsResponse, string) { + t.Helper() + resp, err := http.Get(testServer.URL + path) + if err != nil { + t.Fatalf("GET %s: %v", path, err) + } + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + var out CNPGClusterLogsResponse + if resp.StatusCode == http.StatusOK { + if err := json.Unmarshal(body, &out); err != nil { + t.Fatalf("decode: %v (%s)", err, body) + } + } + return resp.StatusCode, out, string(body) +} + +// useLogServer points Radar's client at an apiserver that serves one JSON log +// line per Pod, so the handler's merge and parse run end to end. +func useLogServer(t *testing.T) { + t.Helper() + apiserver := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + parts := strings.Split(r.URL.Path, "/") + if len(parts) < 2 || parts[len(parts)-1] != "log" { + http.NotFound(w, r) + return + } + pod := parts[len(parts)-2] + fmt.Fprintf(w, "2026-09-28T14:19:58.5Z {\"level\":\"info\",\"logger\":\"postgres\",\"msg\":\"record\",\"record\":{\"error_severity\":\"LOG\",\"message\":\"hello from %s\"}}\n", pod) + })) + t.Cleanup(apiserver.Close) + client, err := kubernetes.NewForConfig(&rest.Config{Host: apiserver.URL}) + if err != nil { + t.Fatal(err) + } + previous := k8s.SetTestClient(client) + t.Cleanup(func() { k8s.SetTestClient(previous) }) +} + +func TestCNPGClusterLogs_OnlyValidatedInstancesContribute(t *testing.T) { + seedCNPGLogCluster(t, "pglogs") + useLogServer(t) + + status, got, body := getCNPGLogs(t, "/api/cnpg/clusters/pglogs/pg-orders/logs") + if status != http.StatusOK { + t.Fatalf("status = %d: %s", status, body) + } + if got.UID != "orders-uid" || got.CapturedAt == "" || got.EmptyMessage == "" { + t.Fatalf("envelope = %+v", got) + } + var names []string + for _, p := range got.Pods { + names = append(names, p.Name) + } + if strings.Join(names, ",") != "pg-orders-1,pg-orders-2" { + t.Fatalf("pods = %v, want only the owned instances", names) + } + for _, entry := range got.Logs { + if entry.Pod != "pg-orders-1" && entry.Pod != "pg-orders-2" { + t.Errorf("log from a non-instance Pod: %+v", entry) + } + if entry.Container != "postgres" { + t.Errorf("container = %q, want the postgres default", entry.Container) + } + } + if len(got.Logs) != 2 { + t.Fatalf("logs = %+v, want one line per instance", got.Logs) + } + for _, entry := range got.Logs { + if entry.Level != "LOG" || entry.Logger != "postgres" || entry.Message != "hello from "+entry.Pod || !strings.HasPrefix(entry.Content, "{") { + t.Errorf("parsed entry = %+v", entry) + } + } + if got.SourceLabels["pg-orders-1"] != "primary" || got.SourceLabels["pg-orders-2"] != "replica" { + t.Errorf("sourceLabels = %v", got.SourceLabels) + } + + status, got, body = getCNPGLogs(t, "/api/cnpg/clusters/pglogs/pg-orders/logs?pod=pg-orders-2") + if status != http.StatusOK || len(got.Pods) != 1 || got.Pods[0].Name != "pg-orders-2" { + t.Fatalf("pod filter: status=%d pods=%+v body=%s", status, got.Pods, body) + } + for _, bad := range []string{"pg-orders-impostor", "pg-orders-orphan", "nope"} { + if status, _, _ := getCNPGLogs(t, "/api/cnpg/clusters/pglogs/pg-orders/logs?pod="+bad); status != http.StatusBadRequest { + t.Errorf("pod=%s: status = %d, want 400", bad, status) + } + } + if status, _, _ := getCNPGLogs(t, "/api/cnpg/clusters/pglogs/pg-orders/logs?sinceTime=yesterday"); status != http.StatusBadRequest { + t.Errorf("bad sinceTime: status = %d, want 400", status) + } +} + +func TestCNPGClusterLogs_NotFound(t *testing.T) { + seedCNPGLogCluster(t, "pglogs404") + for _, path := range []string{ + "/api/cnpg/clusters/pglogs404/missing/logs", + "/api/cnpg/clusters/pglogs404/capi-only/logs", + "/api/cnpg/clusters/elsewhere/pg-orders/logs", + "/api/cnpg/clusters/pglogs404/missing/logs/stream", + } { + if status, _, body := getCNPGLogs(t, path); status != http.StatusNotFound { + t.Errorf("%s: status = %d, want 404 (%s)", path, status, body) + } + } +} + +func TestCNPGClusterLogs_Authorization(t *testing.T) { + seedCNPGLogCluster(t, "pglogsauth") + env := newAuthTestServer(t) + for _, u := range []struct { + name string + clusters bool + }{{"no-clusters", false}, {"no-logs", true}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pglogsauth"}} + perms.SetCanI("get", cnpgGroup, "clusters", "pglogsauth", u.clusters) + allow(perms, "", "pods", "pglogsauth", true) + env.srv.permCache.Set(u.name, nil, perms) + } + for _, tc := range []struct{ user, path, want string }{ + {"no-clusters", "/api/cnpg/clusters/pglogsauth/pg-orders/logs", "clusters.postgresql.cnpg.io"}, + {"no-clusters", "/api/cnpg/clusters/pglogsauth/missing/logs", "clusters.postgresql.cnpg.io"}, + {"no-logs", "/api/cnpg/clusters/pglogsauth/pg-orders/logs", "get pods/log"}, + {"no-logs", "/api/cnpg/clusters/pglogsauth/pg-orders/logs/stream", "get pods/log"}, + } { + resp := env.authGet(t, tc.path, tc.user, "") + body, _ := io.ReadAll(resp.Body) + resp.Body.Close() + if resp.StatusCode != http.StatusForbidden || !strings.Contains(string(body), tc.want) { + t.Errorf("%s %s: status=%d body=%s, want 403 naming %q", tc.user, tc.path, resp.StatusCode, body, tc.want) + } + } +} + +func TestAnnotateCNPGLogEntry(t *testing.T) { + cases := []struct { + name, content, level, logger, message string + }{ + { + name: "postgres record", + content: `{"level":"info","ts":"2026-09-28T14:19:58.123Z","logger":"postgres","msg":"record","record":{"error_severity":"FATAL","message":"password authentication failed","log_time":"2026-09-28 14:19:58.123 UTC"}}`, + level: "FATAL", logger: "postgres", message: "password authentication failed", + }, + { + name: "instance manager error", + content: `{"level":"error","ts":"2026-09-28T14:19:58Z","logger":"barman-cloud-wal-archive","msg":"Error invoking barman-cloud-wal-archive","error":"exit status 4"}`, + level: "ERROR", logger: "barman-cloud-wal-archive", message: "Error invoking barman-cloud-wal-archive: exit status 4", + }, + { + name: "structured error", + content: `{"level":"error","msg":"failed","error":{"code":2}}`, + level: "ERROR", message: `failed: {"code":2}`, + }, + {name: "plain text", content: "LOG: database system is ready"}, + {name: "broken json", content: `{"level":"info"`}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + entry := workloadLogEntry{Content: tc.content} + annotateCNPGLogEntry(&entry) + if entry.Level != tc.level || entry.Logger != tc.logger || entry.Message != tc.message || entry.Content != tc.content { + t.Fatalf("got level=%q logger=%q message=%q content-changed=%v", entry.Level, entry.Logger, entry.Message, entry.Content != tc.content) + } + }) + } + raw, _ := json.Marshal(workloadLogEntry{Pod: "p", Content: "x"}) + if strings.Contains(string(raw), "level") || strings.Contains(string(raw), "message") { + t.Fatalf("unparsed entries grew fields: %s", raw) + } +} + +func TestParseCNPGLogQuery(t *testing.T) { + now := time.Date(2026, 9, 28, 12, 0, 0, 0, time.UTC) + req := httptest.NewRequest("GET", "/?sinceTime=2026-09-28T11:59:00.5Z", nil) + if _, err := parseCNPGLogQuery(req, now); err != nil { + t.Fatalf("fractional RFC3339 rejected: %v", err) + } + req = httptest.NewRequest("GET", "/?sinceTime=2026-09-28T11:58:30Z", nil) + q, err := parseCNPGLogQuery(req, now) + if err != nil || q.sinceSeconds == nil || *q.sinceSeconds != 90 || q.container != "postgres" || q.tailLines != 200 { + t.Fatalf("query = %+v err=%v", q, err) + } + if q.keep(workloadLogEntry{Timestamp: "2026-09-28T11:58:29.9Z"}) || !q.keep(workloadLogEntry{Timestamp: "2026-09-28T11:58:30Z"}) { + t.Fatal("sinceTime overlap not trimmed") + } + req = httptest.NewRequest("GET", "/?sinceTime=2026-09-28T11:58:30Z&sinceSeconds=5", nil) + if _, err := parseCNPGLogQuery(req, now); err == nil { + t.Fatal("sinceTime with sinceSeconds accepted") + } +} + +func useMemoryTimeline(t *testing.T) timeline.EventStore { + t.Helper() + timeline.ResetStore() + if err := timeline.InitStore(timeline.StoreConfig{Type: timeline.StoreTypeMemory, MaxSize: 1000}); err != nil { + t.Fatalf("InitStore: %v", err) + } + t.Cleanup(func() { + timeline.ResetStore() + if err := timeline.InitStore(timeline.DefaultStoreConfig()); err != nil { + t.Fatalf("re-init global store: %v", err) + } + }) + return timeline.GetStore() +} + +type activityRow struct { + id, apiVersion, kind, name, uid string + source timeline.EventSource + eventType timeline.EventType + age time.Duration + labels map[string]string + owner *timeline.OwnerInfo +} + +func seedActivity(t *testing.T, store timeline.EventStore, ns string, rows ...activityRow) { + t.Helper() + now := time.Now() + for _, r := range rows { + source, eventType := r.source, r.eventType + if source == "" { + source = timeline.SourceInformer + } + if eventType == "" { + eventType = timeline.EventTypeUpdate + } + e := timeline.TimelineEvent{ + ID: r.id, Timestamp: now.Add(-r.age), Source: source, Kind: r.kind, APIVersion: r.apiVersion, + Namespace: ns, Name: r.name, UID: r.uid, EventType: eventType, Labels: r.labels, Owner: r.owner, + ClusterContext: k8s.ActiveClusterContext(), + } + if err := store.Append(context.Background(), e); err != nil { + t.Fatalf("append %s: %v", r.id, err) + } + } +} + +func cnpgActivityFixture(t *testing.T, ns string) { + t.Helper() + store := useMemoryTimeline(t) + attributed := map[string]string{pkgtimeline.CNPGClusterLabel: "pg-orders"} + clusterOwner := &timeline.OwnerInfo{Kind: "Cluster", Name: "pg-orders", APIVersion: "postgresql.cnpg.io/v1", UID: "orders-uid"} + seedActivity(t, store, ns, + activityRow{id: "cluster-update", apiVersion: "postgresql.cnpg.io/v1", kind: "Cluster", name: "pg-orders", uid: "orders-uid", age: time.Hour}, + activityRow{id: "backup-add", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-orders-b1", uid: "b1", eventType: timeline.EventTypeAdd, age: 50 * time.Minute, labels: attributed}, + activityRow{id: "backup-delete", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-orders-b1", uid: "b1", eventType: timeline.EventTypeDelete, age: 40 * time.Minute, labels: attributed}, + activityRow{id: "backup-k8s-event", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-orders-b1", uid: "b1", source: timeline.SourceK8sEvent, eventType: timeline.EventTypeWarning, age: 45 * time.Minute}, + activityRow{id: "pod-update", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p1", age: 30 * time.Minute, labels: attributed, owner: clusterOwner}, + activityRow{id: "pod-k8s-event", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p1", source: timeline.SourceK8sEvent, eventType: timeline.EventTypeWarning, age: 20 * time.Minute}, + activityRow{id: "old-pooler", apiVersion: "postgresql.cnpg.io/v1", kind: "Pooler", name: "pg-orders-rw", uid: "pool1", age: 48 * time.Hour, labels: attributed}, + activityRow{id: "other-backup", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-other-b1", uid: "b2", age: 10 * time.Minute, labels: map[string]string{pkgtimeline.CNPGClusterLabel: "pg-other"}}, + activityRow{id: "velero-backup", apiVersion: "velero.io/v1", kind: "Backup", name: "pg-orders", uid: "v1", age: 10 * time.Minute, labels: attributed}, + activityRow{id: "capi-cluster", apiVersion: "cluster.x-k8s.io/v1beta1", kind: "Cluster", name: "pg-orders", uid: "capi", age: 10 * time.Minute}, + activityRow{id: "impostor-pod", apiVersion: "v1", kind: "Pod", name: "impostor", uid: "p9", age: 10 * time.Minute, labels: attributed}, + ) + seedActivity(t, store, "elsewhere", + activityRow{id: "elsewhere-backup", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-orders-b9", uid: "b9", age: 10 * time.Minute, labels: attributed}, + ) +} + +func decodeActivity(t *testing.T, resp *http.Response) CNPGClusterActivityResponse { + t.Helper() + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + if resp.StatusCode != http.StatusOK { + t.Fatalf("status = %d: %s", resp.StatusCode, body) + } + var out CNPGClusterActivityResponse + if err := json.Unmarshal(body, &out); err != nil { + t.Fatalf("decode: %v", err) + } + return out +} + +func activityIDs(resp CNPGClusterActivityResponse) []string { + var out []string + for _, e := range resp.Events { + out = append(out, e.ID) + } + return out +} + +func TestCNPGClusterActivity_AttributesDeletedChildren(t *testing.T) { + cnpgActivityFixture(t, "pgact") + resp, err := http.Get(testServer.URL + "/api/cnpg/clusters/pgact/pg-orders/activity") + if err != nil { + t.Fatal(err) + } + got := decodeActivity(t, resp) + want := "pod-k8s-event,pod-update,backup-delete,backup-k8s-event,backup-add,cluster-update" + if strings.Join(activityIDs(got), ",") != want { + t.Fatalf("events = %v, want %s", activityIDs(got), want) + } + if got.Truncated { + t.Error("truncated on a small history") + } + if got.Oldest == nil || got.AttributionSince == nil { + t.Fatalf("oldest=%v attributionSince=%v", got.Oldest, got.AttributionSince) + } + if age := time.Since(*got.AttributionSince); age < 47*time.Hour { + t.Errorf("attributionSince = %v, want the 48h-old Pooler row outside the window", got.AttributionSince) + } + + resp, _ = http.Get(testServer.URL + "/api/cnpg/clusters/pgact/pg-orders/activity?limit=2&since=" + time.Now().Add(-72*time.Hour).UTC().Format(time.RFC3339)) + got = decodeActivity(t, resp) + if len(got.Events) != 2 || !got.Truncated || got.Events[0].ID != "pod-k8s-event" { + t.Fatalf("limited: events=%v truncated=%v", activityIDs(got), got.Truncated) + } + + for _, q := range []string{"?since=yesterday", "?limit=0", "?limit=x"} { + resp, _ := http.Get(testServer.URL + "/api/cnpg/clusters/pgact/pg-orders/activity" + q) + resp.Body.Close() + if resp.StatusCode != http.StatusBadRequest { + t.Errorf("%s: status = %d, want 400", q, resp.StatusCode) + } + } +} + +func TestCNPGClusterActivity_DropsKindsTheCallerCannotList(t *testing.T) { + cnpgActivityFixture(t, "pgactauth") + env := newAuthTestServer(t) + for _, u := range []struct { + name string + clusterGet bool + backupsListed bool + eventsListed bool + }{{"reader", true, true, true}, {"no-backups", true, false, true}, {"no-clusters", false, true, true}, {"no-events", true, true, false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pgactauth"}} + perms.SetCanI("get", cnpgGroup, "clusters", "pgactauth", u.clusterGet) + allow(perms, "", "events", "pgactauth", u.eventsListed) + allow(perms, cnpgGroup, "clusters", "pgactauth", true) + allow(perms, cnpgGroup, "backups", "pgactauth", u.backupsListed) + allow(perms, cnpgGroup, "poolers", "pgactauth", true) + allow(perms, "", "pods", "pgactauth", true) + env.srv.permCache.Set(u.name, nil, perms) + } + + control := decodeActivity(t, env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "reader", "")) + if !strings.Contains(strings.Join(activityIDs(control), ","), "backup-delete") { + t.Fatalf("control: deleted Backup missing: %v", activityIDs(control)) + } + + got := decodeActivity(t, env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "no-backups", "")) + for _, e := range got.Events { + if e.Kind == "Backup" { + t.Errorf("Backup row reached a caller who cannot list backups: %s", e.ID) + } + } + if len(got.Events) != 3 { + t.Errorf("events = %v, want the Cluster and Pod rows", activityIDs(got)) + } + + sawEvent := false + for _, e := range control.Events { + if e.Source == pkgtimeline.SourceK8sEvent { + sawEvent = true + } + } + if !sawEvent { + t.Fatalf("control: no Kubernetes Event rows: %v", activityIDs(control)) + } + noEvents := decodeActivity(t, env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "no-events", "")) + for _, e := range noEvents.Events { + if e.Source == pkgtimeline.SourceK8sEvent { + t.Errorf("Kubernetes Event row reached a caller who cannot list events: %s", e.ID) + } + } + + resp := env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "no-clusters", "") + resp.Body.Close() + if resp.StatusCode != http.StatusForbidden { + t.Errorf("no cluster get: status = %d, want 403", resp.StatusCode) + } +} + +func TestCNPGClusterLogsStream_SendsParsedInstanceLines(t *testing.T) { + seedCNPGLogCluster(t, "pgstream") + useLogServer(t) + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + req, _ := http.NewRequestWithContext(ctx, "GET", testServer.URL+"/api/cnpg/clusters/pgstream/pg-orders/logs/stream?pod=pg-orders-2", nil) + resp, err := http.DefaultClient.Do(req) + if err != nil { + t.Fatal(err) + } + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK || resp.Header.Get("Content-Type") != "text/event-stream" { + t.Fatalf("status=%d content-type=%q", resp.StatusCode, resp.Header.Get("Content-Type")) + } + buf := make([]byte, 0, 4096) + chunk := make([]byte, 1024) + for !strings.Contains(string(buf), "event: log") { + n, err := resp.Body.Read(chunk) + buf = append(buf, chunk[:n]...) + if err != nil { + t.Fatalf("stream ended before a log event: %v\n%s", err, buf) + } + } + stream := string(buf) + sawConnected := strings.Contains(stream, "event: connected") && strings.Contains(stream, `"name":"pg-orders-2"`) + if !sawConnected || strings.Contains(stream, `"name":"pg-orders-1"`) { + t.Fatalf("connected event wrong:\n%s", stream) + } + for _, want := range []string{`"level":"LOG"`, `"message":"hello from pg-orders-2"`, `"sourceLabel":"replica"`} { + if !strings.Contains(stream, want) { + t.Errorf("stream missing %s:\n%s", want, stream) + } + } +} + +func TestCNPGClusterActivity_RecreatedClusterExcludesPreviousIncarnationPods(t *testing.T) { + owner := func(uid string) *timeline.OwnerInfo { + return &timeline.OwnerInfo{Kind: "Cluster", Name: "pg-orders", APIVersion: "postgresql.cnpg.io/v1", UID: uid} + } + // Seeding the cache records the Cluster's own rows, so the Pod rows are + // seeded into a fresh store afterwards and only Pod rows are compared. + podHistory := func(live ...runtime.Object) []string { + k8s.ResetTestDynamicState() + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, live...) + store := useMemoryTimeline(t) + seedActivity(t, store, "pgrecreate", + activityRow{id: "current-pod", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p-new", age: 10 * time.Minute, owner: owner("orders-uid")}, + activityRow{id: "old-pod", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p-old", age: 2 * time.Hour, owner: owner("previous-uid")}, + activityRow{id: "old-pod-event", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p-old", source: timeline.SourceK8sEvent, eventType: timeline.EventTypeWarning, age: 90 * time.Minute}, + ) + resp, err := http.Get(testServer.URL + "/api/cnpg/clusters/pgrecreate/pg-orders/activity") + if err != nil { + t.Fatal(err) + } + var pods []string + for _, e := range decodeActivity(t, resp).Events { + if e.Kind == "Pod" { + pods = append(pods, e.ID) + } + } + return pods + } + + live := withUID(cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgrecreate", "pg-orders", nil, nil), "orders-uid") + if got := strings.Join(podHistory(live), ","); got != "current-pod" { + t.Fatalf("with the Cluster live: pod events = %s, want only the current incarnation's", got) + } + if got := strings.Join(podHistory(), ","); got != "current-pod,old-pod-event,old-pod" { + t.Fatalf("with the Cluster deleted: pod events = %s, want every incarnation", got) + } +} + +func TestCNPGStreamCursorResumesWithoutReplay(t *testing.T) { + var c cnpgStreamCursor + first := c.restartOptions("postgres", 200, nil) + if first.TailLines == nil || *first.TailLines != 200 || first.SinceTime != nil || !first.Follow || !first.Timestamps || first.Container != "postgres" { + t.Fatalf("first start = %+v", first) + } + line := func(ts, content string) workloadLogEntry { + return workloadLogEntry{Timestamp: ts, Content: content} + } + for _, e := range []workloadLogEntry{line("2026-09-28T14:00:00.1Z", "a"), line("2026-09-28T14:00:05.7Z", "b"), line("2026-09-28T14:00:05.7Z", "c")} { + if !c.admit(e) { + t.Fatalf("fresh line %+v rejected", e) + } + } + + restart := c.restartOptions("postgres", 200, nil) + if restart.TailLines != nil || restart.SinceSeconds != nil || restart.SinceTime == nil || + !restart.SinceTime.Time.Equal(time.Date(2026, 9, 28, 14, 0, 5, 0, time.UTC)) { + t.Fatalf("restart = %+v, want sinceTime at the last delivered second and no tail", restart) + } + + // The resumed follow replays the boundary second. + replayed := []workloadLogEntry{line("2026-09-28T14:00:05.2Z", "earlier in the second"), line("2026-09-28T14:00:05.7Z", "b"), line("2026-09-28T14:00:05.7Z", "c")} + for _, e := range replayed { + if c.admit(e) { + t.Errorf("replayed line %+v admitted", e) + } + } + for _, e := range []workloadLogEntry{line("2026-09-28T14:00:05.7Z", "d"), line("2026-09-28T14:00:06Z", "e")} { + if !c.admit(e) { + t.Errorf("new line %+v rejected", e) + } + } + if c.admit(line("2026-09-28T14:00:05.7Z", "d")) { + t.Error("line before the new last timestamp admitted") + } +} diff --git a/internal/server/cnpg_cluster_logs.go b/internal/server/cnpg_cluster_logs.go new file mode 100644 index 0000000000..19a6436a3c --- /dev/null +++ b/internal/server/cnpg_cluster_logs.go @@ -0,0 +1,567 @@ +package server + +import ( + "bufio" + "context" + "encoding/json" + "errors" + "fmt" + "io" + "log" + "math" + "net/http" + "sort" + "strings" + "sync" + "time" + + "github.com/go-chi/chi/v5" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/labels" + "k8s.io/apimachinery/pkg/types" + "k8s.io/client-go/kubernetes" + + "github.com/skyhook-io/radar/internal/k8s" +) + +const ( + cnpgClusterLabel = "cnpg.io/cluster" + cnpgDefaultLogContainer = "postgres" + cnpgDefaultLogTailLines = 200 + cnpgLogDiscoveryInterval = 5 * time.Second + cnpgLogsEmptyMessage = "No readable logs from this cluster's instances in this snapshot. Refresh after the instances start." + cnpgLogsNoInstanceMessage = "This cluster has no instance Pods yet." +) + +// CNPGClusterLogsResponse is GET /api/cnpg/clusters/{namespace}/{name}/logs. +// Pods and SourceLabels list only the instances that contributed a source to +// this snapshot; SourceLabels maps a Pod to its instance role. +type CNPGClusterLogsResponse struct { + UID types.UID `json:"uid"` + Pods []WorkloadPodInfo `json:"pods"` + Logs []workloadLogEntry `json:"logs"` + Notice string `json:"notice"` + SourceLabels map[string]string `json:"sourceLabels,omitempty"` + CapturedAt string `json:"capturedAt"` + EmptyMessage string `json:"emptyMessage"` +} + +type cnpgLogQuery struct { + container string + tailLines int64 + sinceSeconds *int64 + sinceTime time.Time + pod string +} + +func parseCNPGLogQuery(r *http.Request, now time.Time) (cnpgLogQuery, error) { + q := r.URL.Query() + out := cnpgLogQuery{ + container: q.Get("container"), + tailLines: parseTailLines(q.Get("tailLines"), cnpgDefaultLogTailLines), + sinceSeconds: parseSinceSeconds(q.Get("sinceSeconds")), + pod: q.Get("pod"), + } + if out.container == "" { + out.container = cnpgDefaultLogContainer + } + if raw := q.Get("sinceTime"); raw != "" { + if q.Get("sinceSeconds") != "" { + return out, errors.New("sinceSeconds and sinceTime are mutually exclusive") + } + t, err := time.Parse(time.RFC3339, raw) + if err != nil { + return out, fmt.Errorf("invalid sinceTime %q (expected RFC3339)", raw) + } + out.sinceTime = t + // The pod log API takes whole seconds; round up and trim the overlap + // from the entries afterwards. + secs := max(int64(math.Ceil(now.Sub(t).Seconds())), 1) + out.sinceSeconds = &secs + } + return out, nil +} + +func (q cnpgLogQuery) keep(entry workloadLogEntry) bool { + if q.sinceTime.IsZero() || entry.Timestamp == "" { + return true + } + ts, err := time.Parse(time.RFC3339Nano, entry.Timestamp) + return err != nil || !ts.Before(q.sinceTime) +} + +// authorizeCNPGClusterLogs gates on reading the Cluster, listing its Pods and +// reading their logs — before the Cluster is looked up, so a denied caller +// cannot probe which Clusters exist. +func (s *Server) authorizeCNPGClusterLogs(w http.ResponseWriter, r *http.Request, namespace string) bool { + if !s.requireConnected(w) { + return false + } + if noNamespaceAccess(s.getUserNamespaces(r, []string{namespace})) { + s.writeError(w, http.StatusForbidden, "no access to namespace "+namespace) + return false + } + if !s.canRead(r, cnpgGroup, "clusters", namespace, "get") { + s.writeError(w, http.StatusForbidden, "no access to clusters.postgresql.cnpg.io in namespace "+namespace) + return false + } + if !s.canRead(r, "", "pods", namespace, "list") { + s.writeError(w, http.StatusForbidden, "no access to pods in namespace "+namespace) + return false + } + return s.authorizePodLogRead(w, r, namespace) +} + +// loadCNPGCluster reads one CNPG Cluster from the dynamic cache. The error is +// already written when ok is false. +func (s *Server) loadCNPGCluster(w http.ResponseWriter, r *http.Request, cache *k8s.ResourceCache, namespace, name string) (*unstructured.Unstructured, bool) { + cluster, err := findCNPGCluster(r.Context(), cache, namespace, name) + switch { + case err == nil && cluster != nil: + return cluster, true + case err == nil, errors.Is(err, k8s.ErrUnknownDynamicKind): + s.writeError(w, http.StatusNotFound, "CloudNativePG Cluster "+namespace+"/"+name+" not found") + case errors.Is(err, errDynamicNotSynced): + s.writeError(w, http.StatusServiceUnavailable, "CloudNativePG Clusters are still syncing") + default: + log.Printf("[cnpg] Failed to read Cluster %s/%s: %v", namespace, name, err) + s.writeError(w, http.StatusInternalServerError, "failed to read CloudNativePG Cluster") + } + return nil, false +} + +func findCNPGCluster(ctx context.Context, cache *k8s.ResourceCache, namespace, name string) (*unstructured.Unstructured, error) { + clusters, err := filterCNPGGroup(listDynamicSynced(ctx, cache, "Cluster", cnpgGroup, namespace)) + if err != nil { + return nil, err + } + for _, c := range clusters { + if c.GetNamespace() == namespace && c.GetName() == name && c.GroupVersionKind().Group == cnpgGroup { + return c, nil + } + } + return nil, nil +} + +// cnpgClusterInstancePods returns the Cluster's instance Pods under the same +// label-and-controller-UID rule the workspace uses, sorted by name. +func cnpgClusterInstancePods(cache *k8s.ResourceCache, cluster *unstructured.Unstructured) ([]*corev1.Pod, error) { + lister := cache.Pods() + if lister == nil { + return nil, errors.New("pod cache unavailable") + } + namespace, name := cluster.GetNamespace(), cluster.GetName() + candidates, err := lister.Pods(namespace).List(labels.SelectorFromSet(labels.Set{cnpgClusterLabel: name})) + if err != nil { + return nil, err + } + uids := map[string]types.UID{namespace + "/" + name: cluster.GetUID()} + pods := make([]*corev1.Pod, 0, len(candidates)) + for _, p := range candidates { + if p != nil && isCNPGInstancePod(p, uids) { + pods = append(pods, p) + } + } + sort.Slice(pods, func(i, j int) bool { return pods[i].Name < pods[j].Name }) + return pods, nil +} + +func cnpgInstanceRole(p *corev1.Pod) string { + if role := p.Labels["cnpg.io/instanceRole"]; role != "" { + return role + } + return p.Labels["role"] +} + +// selectCNPGLogPods narrows to the requested instance. ok is false when the +// requested Pod is not one of the Cluster's instances. +func selectCNPGLogPods(pods []*corev1.Pod, want string) ([]*corev1.Pod, bool) { + if want == "" { + return pods, true + } + for _, p := range pods { + if p.Name == want { + return []*corev1.Pod{p}, true + } + } + return nil, false +} + +// cnpgLogRecord is the subset of a CloudNativePG instance-manager JSON log +// line the viewer surfaces. PostgreSQL's own log lines arrive wrapped, with the +// server's severity and message under record. +type cnpgLogRecord struct { + Level string `json:"level"` + Logger string `json:"logger"` + Msg string `json:"msg"` + Error any `json:"error"` + Record *struct { + ErrorSeverity string `json:"error_severity"` + Message string `json:"message"` + } `json:"record"` +} + +// annotateCNPGLogEntry fills the parsed fields of a CloudNativePG log line and +// leaves anything that is not one untouched. +func annotateCNPGLogEntry(entry *workloadLogEntry) { + content := strings.TrimSpace(entry.Content) + if !strings.HasPrefix(content, "{") { + return + } + var rec cnpgLogRecord + if err := json.Unmarshal([]byte(content), &rec); err != nil { + return + } + level := strings.ToUpper(rec.Level) + message := rec.Msg + if rec.Record != nil { + if rec.Record.ErrorSeverity != "" { + level = rec.Record.ErrorSeverity + } + if rec.Record.Message != "" { + message = rec.Record.Message + } + } + if errText := cnpgLogErrorText(rec.Error); errText != "" { + if message == "" { + message = errText + } else { + message += ": " + errText + } + } + entry.Level, entry.Logger, entry.Message = level, rec.Logger, message +} + +func cnpgLogErrorText(v any) string { + switch e := v.(type) { + case nil: + return "" + case string: + return e + default: + b, err := json.Marshal(e) + if err != nil { + return "" + } + return string(b) + } +} + +// handleCNPGClusterLogs serves GET /api/cnpg/clusters/{namespace}/{name}/logs: +// a bounded snapshot of every instance Pod's logs, merged by timestamp. +func (s *Server) handleCNPGClusterLogs(w http.ResponseWriter, r *http.Request) { + namespace, name := chi.URLParam(r, "namespace"), chi.URLParam(r, "name") + if !s.authorizeCNPGClusterLogs(w, r, namespace) { + return + } + query, qerr := parseCNPGLogQuery(r, time.Now()) + if qerr != nil { + s.writeError(w, http.StatusBadRequest, qerr.Error()) + return + } + cache := k8s.GetResourceCache() + if cache == nil { + s.writeError(w, http.StatusServiceUnavailable, "resource cache not available") + return + } + cluster, ok := s.loadCNPGCluster(w, r, cache, namespace, name) + if !ok { + return + } + instances, err := cnpgClusterInstancePods(cache, cluster) + if err != nil { + log.Printf("[cnpg] Failed to list instance Pods for %s/%s: %v", namespace, name, err) + s.writeError(w, http.StatusServiceUnavailable, "instance Pods unavailable: "+err.Error()) + return + } + pods, ok := selectCNPGLogPods(instances, query.pod) + if !ok { + s.writeError(w, http.StatusBadRequest, "pod "+query.pod+" is not an instance of CloudNativePG Cluster "+namespace+"/"+name) + return + } + + resp := CNPGClusterLogsResponse{ + UID: cluster.GetUID(), + Pods: []WorkloadPodInfo{}, + Logs: []workloadLogEntry{}, + CapturedAt: time.Now().UTC().Format(time.RFC3339), + EmptyMessage: cnpgLogsEmptyMessage, + } + if len(pods) == 0 { + resp.EmptyMessage = cnpgLogsNoInstanceMessage + s.writeJSON(w, resp) + return + } + client := s.getClientForRequest(r) + if client == nil { + s.writeError(w, http.StatusServiceUnavailable, "cluster client unavailable") + return + } + + snapshot := collectLogsFromPods(r.Context(), client, namespace, pods, query.container, query.tailLines, query.sinceSeconds, true) + shown := []*corev1.Pod{} + sourceLabels := map[string]string{} + for _, p := range pods { + if !snapshot.SourcePods[p.Name] { + continue + } + shown = append(shown, p) + if role := cnpgInstanceRole(p); role != "" { + sourceLabels[p.Name] = role + } + } + for _, entry := range snapshot.Logs { + if !query.keep(entry) { + continue + } + entry.SourceLabel = sourceLabels[entry.Pod] + annotateCNPGLogEntry(&entry) + resp.Logs = append(resp.Logs, entry) + } + sortLogsByTimestamp(resp.Logs) + resp.Pods = buildPodInfos(shown) + resp.Notice = snapshot.Notice + if len(sourceLabels) > 0 { + resp.SourceLabels = sourceLabels + } + s.writeJSON(w, resp) +} + +// handleCNPGClusterLogsStream serves GET +// /api/cnpg/clusters/{namespace}/{name}/logs/stream: an SSE follow of every +// instance Pod, re-resolving instances as the Cluster fails over or scales. +// Events: connected {cluster, namespace, uid, pods}, log (a log entry with +// the parsed fields), pod_added {pods}, pod_removed {pod, reason}, end +// {reason}, error {error}. +func (s *Server) handleCNPGClusterLogsStream(w http.ResponseWriter, r *http.Request) { + namespace, name := chi.URLParam(r, "namespace"), chi.URLParam(r, "name") + if !s.authorizeCNPGClusterLogs(w, r, namespace) { + return + } + query, qerr := parseCNPGLogQuery(r, time.Now()) + if qerr != nil { + s.writeError(w, http.StatusBadRequest, qerr.Error()) + return + } + cache := k8s.GetResourceCache() + if cache == nil { + s.writeError(w, http.StatusServiceUnavailable, "resource cache not available") + return + } + cluster, ok := s.loadCNPGCluster(w, r, cache, namespace, name) + if !ok { + return + } + instances, err := cnpgClusterInstancePods(cache, cluster) + if err != nil { + log.Printf("[cnpg] Failed to list instance Pods for %s/%s: %v", namespace, name, err) + s.writeError(w, http.StatusServiceUnavailable, "instance Pods unavailable: "+err.Error()) + return + } + pods, ok := selectCNPGLogPods(instances, query.pod) + if !ok { + s.writeError(w, http.StatusBadRequest, "pod "+query.pod+" is not an instance of CloudNativePG Cluster "+namespace+"/"+name) + return + } + client := s.getClientForRequest(r) + if client == nil { + s.writeError(w, http.StatusServiceUnavailable, "cluster client unavailable") + return + } + flusher, ok := w.(http.Flusher) + if !ok { + log.Printf("[cnpg] Failed to stream logs for %s/%s: response writer does not support flushing", namespace, name) + s.writeError(w, http.StatusInternalServerError, "streaming not supported") + return + } + + w.Header().Set("Content-Type", "text/event-stream") + w.Header().Set("Cache-Control", "no-cache") + w.Header().Set("Connection", "keep-alive") + w.Header().Set("X-Accel-Buffering", "no") + + uid := cluster.GetUID() + sendSSEEvent(w, flusher, "connected", map[string]any{ + "cluster": name, "namespace": namespace, "uid": uid, "pods": buildPodInfos(pods), + }) + + ctx, cancel := context.WithCancel(r.Context()) + defer cancel() + logCh := make(chan workloadLogEntry, 1000) + var active sync.Map + roles := map[string]string{} + cursors := map[string]*cnpgStreamCursor{} + start := func(pods []*corev1.Pod) { + for _, pod := range pods { + roles[pod.Name] = cnpgInstanceRole(pod) + for _, c := range k8s.GetContainersForPod(pod, query.container, true) { + key := pod.Name + "/" + c + if _, exists := active.Load(key); exists { + continue + } + cursor := cursors[key] + if cursor == nil { + cursor = &cnpgStreamCursor{} + cursors[key] = cursor + } + opts := cursor.restartOptions(c, query.tailLines, query.sinceSeconds) + streamCtx, streamCancel := context.WithCancel(ctx) + handle := &cnpgStreamHandle{cancel: streamCancel} + active.Store(key, handle) + go func(podName, key string) { + defer active.CompareAndDelete(key, handle) + followCNPGContainerLogs(streamCtx, client, namespace, podName, opts, logCh) + }(pod.Name, key) + } + } + } + start(pods) + + known := map[string]bool{} + for _, p := range pods { + known[p.Name] = true + } + ticker := time.NewTicker(cnpgLogDiscoveryInterval) + defer ticker.Stop() + for { + select { + case <-ctx.Done(): + return + case entry := <-logCh: + if cursor := cursors[entry.Pod+"/"+entry.Container]; cursor != nil && !cursor.admit(entry) { + continue + } + if !query.keep(entry) { + continue + } + entry.SourceLabel = roles[entry.Pod] + annotateCNPGLogEntry(&entry) + sendSSEEvent(w, flusher, "log", entry) + case <-ticker.C: + current, err := findCNPGCluster(ctx, cache, namespace, name) + if err != nil { + continue + } + if current == nil || current.GetUID() != uid { + sendSSEEvent(w, flusher, "end", map[string]string{"reason": "cluster deleted"}) + return + } + all, err := cnpgClusterInstancePods(cache, current) + if err != nil { + continue + } + currentPods, _ := selectCNPGLogPods(all, query.pod) + present := map[string]bool{} + for _, p := range currentPods { + present[p.Name] = true + if !known[p.Name] { + known[p.Name] = true + sendSSEEvent(w, flusher, "pod_added", map[string]any{"pods": []WorkloadPodInfo{buildPodInfo(p, time.Now())}}) + } + } + for podName := range known { + if present[podName] { + continue + } + delete(known, podName) + active.Range(func(key, value any) bool { + if strings.HasPrefix(key.(string), podName+"/") { + value.(*cnpgStreamHandle).cancel() + active.Delete(key) + } + return true + }) + for key := range cursors { + if strings.HasPrefix(key, podName+"/") { + delete(cursors, key) + } + } + sendSSEEvent(w, flusher, "pod_removed", map[string]string{"pod": podName, "reason": "terminated"}) + } + start(currentPods) + } + } +} + +type cnpgStreamHandle struct { + cancel context.CancelFunc +} + +// cnpgStreamCursor remembers where one container's follow left off, so a +// stream that ends while its Pod is still an instance resumes instead of +// replaying lines the client already has. Only the stream loop touches it. +type cnpgStreamCursor struct { + last time.Time + // atLast holds the contents delivered with timestamp == last. The pod log + // API's sinceTime is second-granular, so a resume replays that second and + // only (timestamp, content) tells a replay from a new line. + atLast map[string]bool +} + +// restartOptions returns the follow request for the next (re)start: the +// caller's window the first time, and from the last delivered second after. +func (c *cnpgStreamCursor) restartOptions(container string, tailLines int64, sinceSeconds *int64) corev1.PodLogOptions { + opts := corev1.PodLogOptions{Container: container, Timestamps: true, Follow: true} + if c.last.IsZero() { + opts.TailLines = &tailLines + opts.SinceSeconds = sinceSeconds + return opts + } + since := metav1.NewTime(c.last.Truncate(time.Second)) + opts.SinceTime = &since + return opts +} + +// admit reports whether an entry is new, recording it when it is. Lines +// arrive in order per container, so anything before the last delivered +// timestamp was already sent. +func (c *cnpgStreamCursor) admit(entry workloadLogEntry) bool { + ts, err := time.Parse(time.RFC3339Nano, entry.Timestamp) + if err != nil { + return true + } + switch { + case ts.Before(c.last): + return false + case ts.Equal(c.last): + if c.atLast[entry.Content] { + return false + } + default: + c.last = ts + c.atLast = map[string]bool{} + } + c.atLast[entry.Content] = true + return true +} + +func followCNPGContainerLogs(ctx context.Context, client kubernetes.Interface, namespace, podName string, opts corev1.PodLogOptions, logCh chan<- workloadLogEntry) { + stream, err := client.CoreV1().Pods(namespace).GetLogs(podName, &opts).Stream(ctx) + if err != nil { + if ctx.Err() == nil { + log.Printf("[cnpg] Failed to follow logs for %s/%s/%s: %v", namespace, podName, opts.Container, err) + } + return + } + defer stream.Close() + reader := bufio.NewReader(stream) + for { + line, err := reader.ReadString('\n') + if line = strings.TrimSuffix(line, "\n"); line != "" && (err == nil || err == io.EOF) { + ts, content := parseLogLine(line) + select { + case logCh <- workloadLogEntry{Pod: podName, Container: opts.Container, Timestamp: ts, Content: content}: + case <-ctx.Done(): + return + } + } + if err != nil { + if err != io.EOF && ctx.Err() == nil { + log.Printf("[cnpg] Failed to read logs for %s/%s/%s: %v", namespace, podName, opts.Container, err) + } + return + } + } +} diff --git a/internal/server/cnpg_operator.go b/internal/server/cnpg_operator.go new file mode 100644 index 0000000000..2d48716752 --- /dev/null +++ b/internal/server/cnpg_operator.go @@ -0,0 +1,366 @@ +package server + +import ( + "log" + "net/http" + "sort" + "strings" + + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/labels" + + "github.com/skyhook-io/radar/internal/k8s" +) + +const ( + cnpgOperatorNameLabel = "app.kubernetes.io/name" + cnpgOperatorNameValue = "cloudnative-pg" + cnpgVersionLabel = "app.kubernetes.io/version" + cnpgPluginNameLabel = "cnpg.io/pluginName" + cnpgOperatorContainer = "manager" + cnpgOperatorDeployVar = "OPERATOR_DEPLOYMENT_NAME" + cnpgMonitoringQueriesCM = "MONITORING_QUERIES_CONFIGMAP" + + cnpgOperatorRoleOperator = "operator" + cnpgOperatorRolePlugin = "plugin" + + cnpgConfigPurposeOperator = "operator" + cnpgConfigPurposeMonitoring = "monitoring" +) + +// CNPGOperatorComponent is one operator or plugin Deployment. Version is the +// image tag, else the app.kubernetes.io/version label, else empty. Replica +// counts are nil when unreported, which is not zero. +type CNPGOperatorComponent struct { + Role string `json:"role"` + PluginName string `json:"pluginName,omitempty"` + Namespace string `json:"namespace"` + Deployment string `json:"deployment"` + Image string `json:"image"` + Version string `json:"version"` + ReadyReplicas *int32 `json:"readyReplicas"` + Replicas *int32 `json:"replicas"` +} + +// CNPGOperatorConfigMapState is present only on ConfigMap references. A Secret +// reference never carries it: the endpoint never reads Secrets. +type CNPGOperatorConfigMapState struct { + Exists *bool `json:"exists"` + Readable bool `json:"readable"` + Reason string `json:"reason,omitempty"` + Data map[string]string `json:"data"` +} + +// CNPGOperatorConfigRef is a ConfigMap or Secret the operator is configured +// to read. +type CNPGOperatorConfigRef struct { + Kind string `json:"kind"` + Namespace string `json:"namespace"` + Name string `json:"name"` + Purpose string `json:"purpose"` + *CNPGOperatorConfigMapState +} + +// CNPGOperatorResponse is GET /api/cnpg/operator. +type CNPGOperatorResponse struct { + Coverage map[string]CNPGWorkspaceCoverage `json:"coverage"` + Components []CNPGOperatorComponent `json:"components"` + Config []CNPGOperatorConfigRef `json:"config"` +} + +// handleCNPGOperator serves GET /api/cnpg/operator: the operator and plugin +// Deployments, their versions and readiness, and where the operator's +// configuration lives. +// +// The operator runs in its own namespace (cnpg-system by default) while +// people filter the view to their application namespaces. Following the view +// filter would report "no operator" to anyone looking at their databases, so +// scope follows permission here, as it does for the catalog reverse lookups. +func (s *Server) handleCNPGOperator(w http.ResponseWriter, r *http.Request) { + if !s.requireConnected(w) { + return + } + cache := k8s.GetResourceCache() + if cache == nil { + s.writeError(w, http.StatusServiceUnavailable, "Resource cache not available") + return + } + + scope := s.cnpgOperatorScope(r) + resp := CNPGOperatorResponse{ + Coverage: map[string]CNPGWorkspaceCoverage{}, + Components: []CNPGOperatorComponent{}, + Config: []CNPGOperatorConfigRef{}, + } + + depAcc, depDenied, deployments := s.cnpgOperatorDeployments(r, cache, scope) + resp.Coverage["deployments"] = cnpgCoverageOf(depAcc, depDenied) + svcAcc, svcDenied, services := s.cnpgOperatorServices(r, cache, scope) + resp.Coverage["services"] = cnpgCoverageOf(svcAcc, svcDenied) + + var operators []*appsv1.Deployment + for _, d := range deployments { + if d.Labels[cnpgOperatorNameLabel] == cnpgOperatorNameValue { + operators = append(operators, d) + resp.Components = append(resp.Components, cnpgOperatorComponent(d, cnpgOperatorRoleOperator, "", cnpgOperatorContainerOf(d))) + } + } + + byNamespace := map[string][]*appsv1.Deployment{} + for _, d := range deployments { + byNamespace[d.Namespace] = append(byNamespace[d.Namespace], d) + } + var plugins []CNPGOperatorComponent + for _, svc := range services { + pluginName := svc.Labels[cnpgPluginNameLabel] + if pluginName == "" || !depAcc.covers(svc.Namespace) { + continue + } + matched := false + if len(svc.Spec.Selector) > 0 { + sel := labels.SelectorFromSet(svc.Spec.Selector) + for _, d := range byNamespace[svc.Namespace] { + if sel.Matches(labels.Set(d.Spec.Template.Labels)) { + matched = true + plugins = append(plugins, cnpgOperatorComponent(d, cnpgOperatorRolePlugin, pluginName, firstContainer(d))) + } + } + } + if !matched { + plugins = append(plugins, CNPGOperatorComponent{Role: cnpgOperatorRolePlugin, PluginName: pluginName, Namespace: svc.Namespace}) + } + } + sort.SliceStable(plugins, func(i, j int) bool { + a, b := plugins[i], plugins[j] + if a.PluginName != b.PluginName { + return a.PluginName < b.PluginName + } + if a.Namespace != b.Namespace { + return a.Namespace < b.Namespace + } + return a.Deployment < b.Deployment + }) + resp.Components = append(resp.Components, plugins...) + + resp.Config = s.cnpgOperatorConfig(r, cache, operators) + s.writeJSON(w, resp) +} + +// cnpgOperatorScope is the caller's RBAC scope without the view filter. A +// Radar forced into one namespace still answers only for that namespace. +func (s *Server) cnpgOperatorScope(r *http.Request) []string { + if k8s.ForceNamespaceScope { + target := k8s.GetNamespaceScopeTarget() + if target == "" { + return []string{} + } + return s.getUserNamespaces(r, []string{target}) + } + return s.getUserNamespaces(r, nil) +} + +func (s *Server) cnpgOperatorDeployments(r *http.Request, cache *k8s.ResourceCache, scope []string) (cnpgKindAccess, []string, []*appsv1.Deployment) { + acc, denied, read := s.cnpgTypedScope(r, cache, scope, "apps", "deployments") + if acc.state == cnpgCoverageDenied || acc.state == cnpgCoverageError { + return acc, denied, nil + } + lister := cache.Deployments() + if lister == nil || !cache.IsKindReady("deployments") { + return cnpgKindAccess{state: cnpgCoverageSyncing}, nil, nil + } + var out []*appsv1.Deployment + if read == nil { + out, _ = lister.List(labels.Everything()) + } else { + for _, ns := range read { + items, _ := lister.Deployments(ns).List(labels.Everything()) + out = append(out, items...) + } + } + sort.Slice(out, func(i, j int) bool { + if out[i].Namespace != out[j].Namespace { + return out[i].Namespace < out[j].Namespace + } + return out[i].Name < out[j].Name + }) + return acc, denied, out +} + +func (s *Server) cnpgOperatorServices(r *http.Request, cache *k8s.ResourceCache, scope []string) (cnpgKindAccess, []string, []*corev1.Service) { + acc, denied, read := s.cnpgTypedScope(r, cache, scope, "", "services") + if acc.state == cnpgCoverageDenied || acc.state == cnpgCoverageError { + return acc, denied, nil + } + lister := cache.Services() + if lister == nil || !cache.IsKindReady("services") { + return cnpgKindAccess{state: cnpgCoverageSyncing}, nil, nil + } + hasPlugin, err := labels.Parse(cnpgPluginNameLabel) + if err != nil { + log.Printf("[cnpg] Failed to build plugin selector: %v", err) + return cnpgKindAccess{state: cnpgCoverageError}, nil, nil + } + var out []*corev1.Service + if read == nil { + out, _ = lister.List(hasPlugin) + } else { + for _, ns := range read { + items, _ := lister.Services(ns).List(hasPlugin) + out = append(out, items...) + } + } + return acc, denied, out +} + +func cnpgOperatorContainerOf(d *appsv1.Deployment) *corev1.Container { + for i := range d.Spec.Template.Spec.Containers { + if d.Spec.Template.Spec.Containers[i].Name == cnpgOperatorContainer { + return &d.Spec.Template.Spec.Containers[i] + } + } + return firstContainer(d) +} + +func firstContainer(d *appsv1.Deployment) *corev1.Container { + if len(d.Spec.Template.Spec.Containers) == 0 { + return nil + } + return &d.Spec.Template.Spec.Containers[0] +} + +func cnpgOperatorComponent(d *appsv1.Deployment, role, pluginName string, c *corev1.Container) CNPGOperatorComponent { + out := CNPGOperatorComponent{ + Role: role, + PluginName: pluginName, + Namespace: d.Namespace, + Deployment: d.Name, + Replicas: d.Spec.Replicas, + } + if c != nil { + out.Image = c.Image + out.Version = imageTag(c.Image) + } + if out.Version == "" { + out.Version = d.Labels[cnpgVersionLabel] + } + if out.Version == "" { + out.Version = d.Spec.Template.Labels[cnpgVersionLabel] + } + // The typed status cannot tell an omitted readyReplicas from zero; a status + // the controller has observed at least once states it authoritatively. + if d.Status.ObservedGeneration > 0 { + ready := d.Status.ReadyReplicas + out.ReadyReplicas = &ready + } + return out +} + +// cnpgOperatorArg returns the value of --flag=value or --flag value from a +// container's command and args. +func cnpgOperatorArg(c *corev1.Container, flag string) string { + argv := append(append([]string{}, c.Command...), c.Args...) + for i, a := range argv { + if v, ok := strings.CutPrefix(a, flag+"="); ok { + return v + } + if a == flag && i+1 < len(argv) { + return argv[i+1] + } + } + return "" +} + +func cnpgOperatorEnv(c *corev1.Container, name string) string { + for _, e := range c.Env { + if e.Name == name && e.ValueFrom == nil { + return e.Value + } + } + return "" +} + +// cnpgOperatorExpand resolves $(OPERATOR_DEPLOYMENT_NAME) the way the kubelet +// would: from the container's literal env, which the shipped manifests set to +// the Deployment's own name. Any other reference is left verbatim, as the +// kubelet leaves an unresolvable one. +func cnpgOperatorExpand(v string, c *corev1.Container, d *appsv1.Deployment) string { + ref := "$(" + cnpgOperatorDeployVar + ")" + if !strings.Contains(v, ref) { + return v + } + name := cnpgOperatorEnv(c, cnpgOperatorDeployVar) + if name == "" { + name = d.Name + } + return strings.ReplaceAll(v, ref, name) +} + +func (s *Server) cnpgOperatorConfig(r *http.Request, cache *k8s.ResourceCache, operators []*appsv1.Deployment) []CNPGOperatorConfigRef { + out := []CNPGOperatorConfigRef{} + seen := map[string]bool{} + add := func(ref CNPGOperatorConfigRef) { + key := ref.Kind + "\x00" + ref.Namespace + "\x00" + ref.Name + "\x00" + ref.Purpose + if ref.Name == "" || seen[key] { + return + } + seen[key] = true + out = append(out, ref) + } + for _, d := range operators { + c := cnpgOperatorContainerOf(d) + if c == nil { + continue + } + if name := cnpgOperatorExpand(cnpgOperatorArg(c, "--config-map-name"), c, d); name != "" { + add(s.cnpgOperatorConfigMap(r, cache, d.Namespace, name, cnpgConfigPurposeOperator)) + } + if name := cnpgOperatorExpand(cnpgOperatorArg(c, "--secret-name"), c, d); name != "" { + add(CNPGOperatorConfigRef{Kind: "Secret", Namespace: d.Namespace, Name: name, Purpose: cnpgConfigPurposeOperator}) + } + if name := cnpgOperatorEnv(c, cnpgMonitoringQueriesCM); name != "" { + add(s.cnpgOperatorConfigMap(r, cache, d.Namespace, name, cnpgConfigPurposeMonitoring)) + } + } + return out +} + +func (s *Server) cnpgOperatorConfigMap(r *http.Request, cache *k8s.ResourceCache, namespace, name, purpose string) CNPGOperatorConfigRef { + ref := CNPGOperatorConfigRef{Kind: "ConfigMap", Namespace: namespace, Name: name, Purpose: purpose} + state := &CNPGOperatorConfigMapState{} + ref.CNPGOperatorConfigMapState = state + if !s.canRead(r, "", "configmaps", namespace, "get") { + state.Reason = "no permission to get ConfigMaps in " + namespace + return ref + } + lister := cache.ConfigMaps() + if lister == nil { + state.Reason = "ConfigMaps are still loading" + return ref + } + if !capacityCacheCoversNamespace(cache, "configmaps", namespace) { + state.Reason = "Radar does not watch ConfigMaps in " + namespace + return ref + } + cm, err := lister.ConfigMaps(namespace).Get(name) + switch { + case apierrors.IsNotFound(err): + exists := false + state.Exists = &exists + state.Reason = "not found" + return ref + case err != nil: + log.Printf("[cnpg] Failed to read ConfigMap %s/%s: %v", namespace, name, err) + state.Reason = "could not read the ConfigMap" + return ref + } + exists := true + state.Exists = &exists + state.Readable = true + state.Data = map[string]string{} + for k, v := range cm.Data { + state.Data[k] = v + } + return ref +} diff --git a/internal/server/cnpg_operator_test.go b/internal/server/cnpg_operator_test.go new file mode 100644 index 0000000000..b1f8e49af6 --- /dev/null +++ b/internal/server/cnpg_operator_test.go @@ -0,0 +1,368 @@ +package server + +import ( + "context" + "encoding/json" + "io" + "net/http" + "testing" + "time" + + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + "github.com/skyhook-io/radar/internal/auth" + "github.com/skyhook-io/radar/internal/k8s" +) + +func cnpgOperatorDeployment() *appsv1.Deployment { + return &appsv1.Deployment{ + ObjectMeta: metav1.ObjectMeta{ + Name: "cnpg-controller-manager", Namespace: "cnpg-system", Generation: 1, + Labels: map[string]string{"app.kubernetes.io/name": "cloudnative-pg"}, + }, + Spec: appsv1.DeploymentSpec{ + Replicas: int32Ptr(1), + Selector: &metav1.LabelSelector{MatchLabels: map[string]string{"app.kubernetes.io/name": "cloudnative-pg"}}, + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{"app.kubernetes.io/name": "cloudnative-pg"}}, + Spec: corev1.PodSpec{Containers: []corev1.Container{ + {Name: "sidecar", Image: "busybox:1.36"}, + { + Name: "manager", + Image: "ghcr.io/cloudnative-pg/cloudnative-pg:1.27.0", + Command: []string{"/manager"}, + Args: []string{ + "controller", "--leader-elect", + "--config-map-name=$(OPERATOR_DEPLOYMENT_NAME)-config", + "--secret-name", "$(OPERATOR_DEPLOYMENT_NAME)-config", + }, + Env: []corev1.EnvVar{ + {Name: "OPERATOR_DEPLOYMENT_NAME", Value: "cnpg-controller-manager"}, + {Name: "MONITORING_QUERIES_CONFIGMAP", Value: "cnpg-default-monitoring"}, + }, + }, + }}, + }, + }, + Status: appsv1.DeploymentStatus{ObservedGeneration: 1, Replicas: 1, ReadyReplicas: 1}, + } +} + +func cnpgPluginDeployment() *appsv1.Deployment { + return &appsv1.Deployment{ + ObjectMeta: metav1.ObjectMeta{Name: "barman-cloud", Namespace: "cnpg-system"}, + Spec: appsv1.DeploymentSpec{ + Selector: &metav1.LabelSelector{MatchLabels: map[string]string{"app": "barman-cloud"}}, + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{"app": "barman-cloud"}}, + Spec: corev1.PodSpec{Containers: []corev1.Container{ + {Name: "barman-cloud", Image: "ghcr.io/cloudnative-pg/plugin-barman-cloud:v0.5.0"}, + }}, + }, + }, + } +} + +func cnpgPluginService() *corev1.Service { + return &corev1.Service{ + ObjectMeta: metav1.ObjectMeta{ + Name: "barman-cloud", Namespace: "cnpg-system", + Labels: map[string]string{"cnpg.io/pluginName": "barman-cloud.cloudnative-pg.io"}, + }, + Spec: corev1.ServiceSpec{Selector: map[string]string{"app": "barman-cloud"}}, + } +} + +func cnpgOperatorConfigMapObj() *corev1.ConfigMap { + return &corev1.ConfigMap{ + ObjectMeta: metav1.ObjectMeta{Name: "cnpg-controller-manager-config", Namespace: "cnpg-system"}, + Data: map[string]string{"INHERITED_ANNOTATIONS": "team/*"}, + } +} + +// seedCNPGOperator creates typed objects in the shared fake cluster and waits +// until the cache serves each one. +func seedCNPGOperator(t *testing.T, deployments []*appsv1.Deployment, services []*corev1.Service, configMaps []*corev1.ConfigMap) { + t.Helper() + ctx := context.Background() + for _, d := range deployments { + if _, err := testFakeClient.AppsV1().Deployments(d.Namespace).Create(ctx, d, metav1.CreateOptions{}); err != nil { + t.Fatalf("create deployment %s: %v", d.Name, err) + } + t.Cleanup(func() { + _ = testFakeClient.AppsV1().Deployments(d.Namespace).Delete(context.Background(), d.Name, metav1.DeleteOptions{}) + }) + } + for _, svc := range services { + if _, err := testFakeClient.CoreV1().Services(svc.Namespace).Create(ctx, svc, metav1.CreateOptions{}); err != nil { + t.Fatalf("create service %s: %v", svc.Name, err) + } + t.Cleanup(func() { + _ = testFakeClient.CoreV1().Services(svc.Namespace).Delete(context.Background(), svc.Name, metav1.DeleteOptions{}) + }) + } + for _, cm := range configMaps { + if _, err := testFakeClient.CoreV1().ConfigMaps(cm.Namespace).Create(ctx, cm, metav1.CreateOptions{}); err != nil { + t.Fatalf("create configmap %s: %v", cm.Name, err) + } + t.Cleanup(func() { + _ = testFakeClient.CoreV1().ConfigMaps(cm.Namespace).Delete(context.Background(), cm.Name, metav1.DeleteOptions{}) + }) + } + cache := k8s.GetResourceCache() + deadline := time.Now().Add(5 * time.Second) + for { + missing := 0 + for _, d := range deployments { + if _, err := cache.Deployments().Deployments(d.Namespace).Get(d.Name); err != nil { + missing++ + } + } + for _, svc := range services { + if _, err := cache.Services().Services(svc.Namespace).Get(svc.Name); err != nil { + missing++ + } + } + for _, cm := range configMaps { + if l := cache.ConfigMaps(); l == nil { + missing++ + } else if _, err := l.ConfigMaps(cm.Namespace).Get(cm.Name); err != nil { + missing++ + } + } + if missing == 0 { + return + } + if time.Now().After(deadline) { + t.Fatalf("%d operator fixtures did not reach the cache", missing) + } + time.Sleep(20 * time.Millisecond) + } +} + +func seedFullCNPGOperator(t *testing.T) { + t.Helper() + seedCNPGOperator(t, + []*appsv1.Deployment{cnpgOperatorDeployment(), cnpgPluginDeployment()}, + []*corev1.Service{cnpgPluginService()}, + []*corev1.ConfigMap{cnpgOperatorConfigMapObj()}, + ) +} + +func readCNPGOperator(t *testing.T, resp *http.Response) (CNPGOperatorResponse, []byte) { + t.Helper() + defer resp.Body.Close() + body, err := io.ReadAll(resp.Body) + if err != nil { + t.Fatalf("read body: %v", err) + } + if resp.StatusCode != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", resp.StatusCode, body) + } + var out CNPGOperatorResponse + if err := json.Unmarshal(body, &out); err != nil { + t.Fatalf("decode: %v", err) + } + return out, body +} + +func getCNPGOperatorNoAuth(t *testing.T, query string) (CNPGOperatorResponse, []byte) { + t.Helper() + resp, err := http.Get(testServer.URL + "/api/cnpg/operator" + query) + if err != nil { + t.Fatalf("GET: %v", err) + } + return readCNPGOperator(t, resp) +} + +func findConfigRef(refs []CNPGOperatorConfigRef, kind, purpose string) *CNPGOperatorConfigRef { + for i := range refs { + if refs[i].Kind == kind && refs[i].Purpose == purpose { + return &refs[i] + } + } + return nil +} + +func TestCNPGOperator_DiscoversOperatorPluginAndConfig(t *testing.T) { + seedFullCNPGOperator(t) + + got, body := getCNPGOperatorNoAuth(t, "") + for _, key := range []string{"deployments", "services"} { + if got.Coverage[key].State != cnpgCoverageFull { + t.Errorf("coverage[%s] = %+v, want full", key, got.Coverage[key]) + } + } + if len(got.Components) != 2 { + t.Fatalf("components = %+v, want operator then plugin", got.Components) + } + op, plugin := got.Components[0], got.Components[1] + if op.Role != "operator" || op.Namespace != "cnpg-system" || op.Deployment != "cnpg-controller-manager" || + op.Image != "ghcr.io/cloudnative-pg/cloudnative-pg:1.27.0" || op.Version != "1.27.0" { + t.Errorf("operator = %+v", op) + } + if op.ReadyReplicas == nil || *op.ReadyReplicas != 1 || op.Replicas == nil || *op.Replicas != 1 { + t.Errorf("operator readiness = %v/%v, want 1/1", op.ReadyReplicas, op.Replicas) + } + if plugin.Role != "plugin" || plugin.PluginName != "barman-cloud.cloudnative-pg.io" || plugin.Deployment != "barman-cloud" || plugin.Version != "v0.5.0" { + t.Errorf("plugin = %+v", plugin) + } + if plugin.ReadyReplicas != nil { + t.Errorf("plugin readyReplicas = %d, want null when the controller has reported no status", *plugin.ReadyReplicas) + } + + cm := findConfigRef(got.Config, "ConfigMap", "operator") + if cm == nil || cm.Name != "cnpg-controller-manager-config" || cm.Namespace != "cnpg-system" || cm.CNPGOperatorConfigMapState == nil || + !cm.Readable || cm.Exists == nil || !*cm.Exists || cm.Data["INHERITED_ANNOTATIONS"] != "team/*" { + t.Errorf("operator ConfigMap = %+v", cm) + } + secret := findConfigRef(got.Config, "Secret", "operator") + if secret == nil || secret.Name != "cnpg-controller-manager-config" { + t.Errorf("operator Secret = %+v, want the space-separated --secret-name resolved", secret) + } + mon := findConfigRef(got.Config, "ConfigMap", "monitoring") + if mon == nil || mon.Name != "cnpg-default-monitoring" || mon.Readable || mon.Exists == nil || *mon.Exists { + t.Errorf("monitoring ConfigMap = %+v, want exists=false readable=false", mon) + } + + var raw struct { + Config []map[string]any `json:"config"` + Components []map[string]any `json:"components"` + } + if err := json.Unmarshal(body, &raw); err != nil { + t.Fatalf("raw decode: %v", err) + } + for _, ref := range raw.Config { + if ref["kind"] != "Secret" { + continue + } + for _, k := range []string{"data", "exists", "readable", "keys"} { + if _, ok := ref[k]; ok { + t.Errorf("Secret reference carries %q: %v", k, ref) + } + } + } + if v, ok := raw.Components[1]["readyReplicas"]; !ok || v != nil { + t.Errorf("plugin readyReplicas JSON = %v (present=%v), want explicit null", v, ok) + } +} + +func TestCNPGOperator_VersionFallsBackToLabelAndNeverInvents(t *testing.T) { + digest := cnpgOperatorDeployment() + digest.Name = "pinned" + digest.Labels["app.kubernetes.io/version"] = "1.26.1" + digest.Spec.Template.Spec.Containers[1].Image = "ghcr.io/cloudnative-pg/cloudnative-pg@sha256:abc" + bare := cnpgOperatorDeployment() + bare.Name = "bare" + bare.Spec.Template.Spec.Containers[1].Image = "ghcr.io/cloudnative-pg/cloudnative-pg" + seedCNPGOperator(t, []*appsv1.Deployment{digest, bare}, nil, nil) + + got, _ := getCNPGOperatorNoAuth(t, "") + versions := map[string]string{} + for _, c := range got.Components { + versions[c.Deployment] = c.Version + } + if versions["pinned"] != "1.26.1" { + t.Errorf("digest-pinned version = %q, want the version label", versions["pinned"]) + } + if v, ok := versions["bare"]; !ok || v != "" { + t.Errorf("untagged, unlabelled version = %q (found=%v), want empty", v, ok) + } +} + +func TestCNPGOperator_IgnoresNamespaceViewFilter(t *testing.T) { + seedFullCNPGOperator(t) + got, _ := getCNPGOperatorNoAuth(t, "?namespaces=default") + if len(got.Components) == 0 || got.Components[0].Deployment != "cnpg-controller-manager" { + t.Errorf("components = %+v, want the operator in cnpg-system despite a view filter on default", got.Components) + } + + env := newAuthTestServer(t) + perms := &auth.UserPermissions{AllowedNamespaces: []string{"cnpg-system", "default"}} + allow(perms, "apps", "deployments", "", true) + allow(perms, "", "services", "", true) + env.srv.permCache.Set("viewer", nil, perms) + authed, _ := readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator?namespaces=default", "viewer", "")) + if len(authed.Components) == 0 || authed.Components[0].Namespace != "cnpg-system" { + t.Errorf("auth components = %+v, want the operator despite the view filter", authed.Components) + } +} + +func TestCNPGOperator_ConfigMapDataNeedsGet(t *testing.T) { + seedFullCNPGOperator(t) + env := newAuthTestServer(t) + for _, u := range []struct { + name string + getCM bool + }{{"reads-cm", true}, {"no-cm", false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"cnpg-system"}} + allow(perms, "apps", "deployments", "", true) + allow(perms, "", "services", "", true) + perms.SetCanI("get", "", "configmaps", "cnpg-system", u.getCM) + env.srv.permCache.Set(u.name, nil, perms) + } + + got, _ := readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator", "reads-cm", "")) + cm := findConfigRef(got.Config, "ConfigMap", "operator") + if cm == nil || cm.CNPGOperatorConfigMapState == nil || !cm.Readable || cm.Data["INHERITED_ANNOTATIONS"] != "team/*" { + t.Errorf("with get configmaps: %+v", cm) + } + + got, body := readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator", "no-cm", "")) + cm = findConfigRef(got.Config, "ConfigMap", "operator") + if cm == nil || cm.CNPGOperatorConfigMapState == nil || cm.Readable || cm.Exists != nil || cm.Data != nil || cm.Reason == "" { + t.Errorf("without get configmaps: %+v, want unreadable, existence unknown, no data, a reason", cm) + } + var raw struct { + Config []map[string]any `json:"config"` + } + _ = json.Unmarshal(body, &raw) + for _, ref := range raw.Config { + if ref["kind"] == "ConfigMap" && ref["purpose"] == "operator" && ref["data"] != nil { + t.Errorf("ConfigMap data returned without get: %v", ref) + } + } +} + +func TestCNPGOperator_DeniedDeploymentsWithholdComponents(t *testing.T) { + seedFullCNPGOperator(t) + env := newAuthTestServer(t) + + partial := &auth.UserPermissions{AllowedNamespaces: []string{"cnpg-system", "default"}} + allow(partial, "apps", "deployments", "", false) + allow(partial, "apps", "deployments", "cnpg-system", false) + allow(partial, "apps", "deployments", "default", true) + allow(partial, "", "services", "", true) + env.srv.permCache.Set("partial", nil, partial) + + got, _ := readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator", "partial", "")) + cov := got.Coverage["deployments"] + if cov.State != cnpgCoveragePartial || len(cov.DeniedNamespaces) != 1 || cov.DeniedNamespaces[0] != "cnpg-system" { + t.Errorf("deployments coverage = %+v, want partial denied [cnpg-system]", cov) + } + for _, c := range got.Components { + if c.Namespace == "cnpg-system" { + t.Errorf("component from a namespace whose Deployments are denied: %+v", c) + } + } + if len(got.Config) != 0 { + t.Errorf("config = %+v, want none without a visible operator", got.Config) + } + + none := &auth.UserPermissions{AllowedNamespaces: []string{"cnpg-system"}} + allow(none, "apps", "deployments", "", false) + allow(none, "apps", "deployments", "cnpg-system", false) + allow(none, "", "services", "", false) + allow(none, "", "services", "cnpg-system", false) + env.srv.permCache.Set("none", nil, none) + + got, _ = readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator", "none", "")) + if got.Coverage["deployments"].State != cnpgCoverageDenied || got.Coverage["services"].State != cnpgCoverageDenied { + t.Errorf("coverage = %+v, want both denied", got.Coverage) + } + if got.Components == nil || len(got.Components) != 0 || got.Config == nil { + t.Errorf("components=%v config=%v, want empty arrays", got.Components, got.Config) + } +} diff --git a/internal/server/cnpg_workspace.go b/internal/server/cnpg_workspace.go new file mode 100644 index 0000000000..1421e0218a --- /dev/null +++ b/internal/server/cnpg_workspace.go @@ -0,0 +1,677 @@ +package server + +import ( + "context" + "errors" + "log" + "net/http" + "slices" + "sort" + "time" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/apimachinery/pkg/types" + + "github.com/skyhook-io/radar/internal/issues" + "github.com/skyhook-io/radar/internal/k8s" + bp "github.com/skyhook-io/radar/pkg/audit" + "github.com/skyhook-io/radar/pkg/issuesapi" +) + +const cnpgBarmanGroup = "barmancloud.cnpg.io" + +const ( + cnpgCoverageFull = "full" + cnpgCoveragePartial = "partial" + cnpgCoverageDenied = "denied" + cnpgCoverageNotInstalled = "notInstalled" + cnpgCoverageSyncing = "syncing" + cnpgCoverageError = "error" +) + +const ( + cnpgWorkspacePodsKey = "pods" + cnpgWorkspaceBackupsKey = "backups" + cnpgWorkspaceSchedKey = "scheduledBackups" + cnpgWorkspaceClusterKey = "clusters" +) + +// cnpgBackupWindow bounds how far back settled Backups are returned. The newest +// completed Backup per Cluster is kept regardless: it is the last-good-backup +// fact the workspace reports. +const cnpgBackupWindow = 7 * 24 * time.Hour + +const cnpgNoDeclarativeBackupCheckID = "cnpgNoDeclarativeBackup" + +type cnpgWorkspaceKind struct { + key string + group string + kind string + resource string + clusterScoped bool +} + +var cnpgWorkspaceKinds = []cnpgWorkspaceKind{ + {key: cnpgWorkspaceClusterKey, group: cnpgGroup, kind: "Cluster", resource: "clusters"}, + {key: cnpgWorkspaceBackupsKey, group: cnpgGroup, kind: "Backup", resource: "backups"}, + {key: cnpgWorkspaceSchedKey, group: cnpgGroup, kind: "ScheduledBackup", resource: "scheduledbackups"}, + {key: "poolers", group: cnpgGroup, kind: "Pooler", resource: "poolers"}, + {key: "databases", group: cnpgGroup, kind: "Database", resource: "databases"}, + {key: "publications", group: cnpgGroup, kind: "Publication", resource: "publications"}, + {key: "subscriptions", group: cnpgGroup, kind: "Subscription", resource: "subscriptions"}, + {key: "imageCatalogs", group: cnpgGroup, kind: "ImageCatalog", resource: "imagecatalogs"}, + {key: "clusterImageCatalogs", group: cnpgGroup, kind: "ClusterImageCatalog", resource: "clusterimagecatalogs", clusterScoped: true}, + {key: "objectStores", group: cnpgBarmanGroup, kind: "ObjectStore", resource: "objectstores"}, +} + +// CNPGWorkspaceCoverage states how much of one kind the caller could see. +// DeniedNamespaces lists only namespaces already in the caller's scope, so it +// may be omitted on a partial state; AllowedNamespaces is always set on a +// partial state and is the authority for which namespaces were read. +type CNPGWorkspaceCoverage struct { + State string `json:"state"` + DeniedNamespaces []string `json:"deniedNamespaces,omitempty"` + AllowedNamespaces []string `json:"allowedNamespaces,omitempty"` +} + +func cnpgCoverageOf(acc cnpgKindAccess, denied []string) CNPGWorkspaceCoverage { + cov := CNPGWorkspaceCoverage{State: acc.state, DeniedNamespaces: denied} + if acc.state == cnpgCoveragePartial { + cov.AllowedNamespaces = make([]string, 0, len(acc.namespaces)) + for ns := range acc.namespaces { + cov.AllowedNamespaces = append(cov.AllowedNamespaces, ns) + } + sort.Strings(cov.AllowedNamespaces) + } + return cov +} + +// CNPGWorkspaceIssue is the subset of issuesapi.Issue the workspace renders. +type CNPGWorkspaceIssue struct { + ID string `json:"id"` + Severity issuesapi.Severity `json:"severity"` + Category issuesapi.Category `json:"category"` + Kind string `json:"kind"` + Group string `json:"group,omitempty"` + Namespace string `json:"namespace,omitempty"` + Name string `json:"name"` + Reason string `json:"reason"` + Message string `json:"message,omitempty"` + Cause string `json:"cause,omitempty"` + Action string `json:"action,omitempty"` + FirstSeen time.Time `json:"first_seen,omitzero"` +} + +// CNPGWorkspaceAuditFinding is one audit finding on a visible CNPG object. +type CNPGWorkspaceAuditFinding struct { + CheckID string `json:"checkId"` + Severity string `json:"severity"` + Kind string `json:"kind"` + Group string `json:"group,omitempty"` + Namespace string `json:"namespace"` + Name string `json:"name"` + Message string `json:"message"` +} + +// CNPGWorkspaceResponse is GET /api/cnpg/workspace. +type CNPGWorkspaceResponse struct { + Installed bool `json:"installed"` + Context string `json:"context"` + Namespaces []string `json:"namespaces"` + Coverage map[string]CNPGWorkspaceCoverage `json:"coverage"` + Objects map[string][]any `json:"objects"` + Issues []CNPGWorkspaceIssue `json:"issues"` + Audit []CNPGWorkspaceAuditFinding `json:"audit"` + BackupsOmitted int `json:"backupsOmitted"` +} + +// cnpgKindAccess is the resolved read scope for one kind. all means every +// namespace in the request's scope (or the cluster-scoped kind itself). +type cnpgKindAccess struct { + state string + all bool + namespaces map[string]bool +} + +func (a cnpgKindAccess) covers(namespace string) bool { + if a.state != cnpgCoverageFull && a.state != cnpgCoveragePartial { + return false + } + return a.all || a.namespaces[namespace] +} + +func newCNPGWorkspaceResponse(namespaces []string) CNPGWorkspaceResponse { + resp := CNPGWorkspaceResponse{ + Context: k8s.ActiveClusterContext(), + Namespaces: namespaces, + Coverage: map[string]CNPGWorkspaceCoverage{}, + Objects: map[string][]any{}, + Issues: []CNPGWorkspaceIssue{}, + Audit: []CNPGWorkspaceAuditFinding{}, + } + for _, k := range cnpgWorkspaceKinds { + resp.Coverage[k.key] = CNPGWorkspaceCoverage{State: cnpgCoverageNotInstalled} + resp.Objects[k.key] = []any{} + } + resp.Coverage[cnpgWorkspacePodsKey] = CNPGWorkspaceCoverage{State: cnpgCoverageNotInstalled} + resp.Objects[cnpgWorkspacePodsKey] = []any{} + return resp +} + +// handleCNPGWorkspace serves GET /api/cnpg/workspace: every CloudNativePG kind +// plus instance Pods, each authorized on its own. The generic resource list +// does not gate namespaced CRDs per kind, so it cannot tell "no access" from +// "none"; this endpoint states which one it is for every kind. +func (s *Server) handleCNPGWorkspace(w http.ResponseWriter, r *http.Request) { + if !s.requireConnected(w) { + return + } + cache := k8s.GetResourceCache() + if cache == nil { + s.writeError(w, http.StatusServiceUnavailable, "Resource cache not available") + return + } + + namespaces := s.parseNamespacesForUser(r) + resp := newCNPGWorkspaceResponse(namespaces) + + disc := k8s.GetResourceDiscovery() + if disc != nil { + for _, k := range cnpgWorkspaceKinds { + if _, ok := disc.GetGVRWithGroup(k.kind, k.group); ok { + resp.Installed = true + break + } + } + if !resp.Installed { + s.writeJSON(w, resp) + return + } + } + + access := map[string]cnpgKindAccess{} + items := map[string][]*unstructured.Unstructured{} + for _, k := range cnpgWorkspaceKinds { + if disc != nil { + if _, ok := disc.GetGVRWithGroup(k.kind, k.group); !ok { + access[k.key] = cnpgKindAccess{state: cnpgCoverageNotInstalled} + continue + } + } + acc, denied, list := s.cnpgWorkspaceReadKind(r, cache, k, namespaces) + if acc.state != cnpgCoverageNotInstalled { + resp.Installed = true + } + access[k.key] = acc + items[k.key] = list + resp.Coverage[k.key] = cnpgCoverageOf(acc, denied) + } + if !resp.Installed { + s.writeJSON(w, resp) + return + } + + for _, k := range cnpgWorkspaceKinds { + list := items[k.key] + if k.key == cnpgWorkspaceBackupsKey { + var omitted int + list, omitted = windowCNPGBackups(list, time.Now()) + resp.BackupsOmitted = omitted + } else { + sortCNPGObjects(list) + } + out := make([]any, 0, len(list)) + for _, u := range list { + out = append(out, u.Object) + } + resp.Objects[k.key] = out + } + + podAccess, podDenied, pods, instancePods := s.cnpgWorkspaceReadPods(r, cache, namespaces, cnpgClusterUIDs(items[cnpgWorkspaceClusterKey])) + access[cnpgWorkspacePodsKey] = podAccess + resp.Coverage[cnpgWorkspacePodsKey] = cnpgCoverageOf(podAccess, podDenied) + resp.Objects[cnpgWorkspacePodsKey] = pods + + resp.Issues = s.cnpgWorkspaceIssues(r, namespaces, access, instancePods) + resp.Audit = cnpgWorkspaceAudit(items[cnpgWorkspaceClusterKey], items[cnpgWorkspaceSchedKey], access[cnpgWorkspaceSchedKey]) + + s.writeJSON(w, resp) +} + +// cnpgWorkspaceScope resolves where the caller may list one namespaced +// resource: nil allowed means the whole request scope. +// +// denied names namespaces only when the candidate set came from the caller — +// their view filter or their RBAC-allowed list. When the scope is "all" the +// candidates are every namespace in Radar's cache, and naming the denied ones +// would disclose namespaces the caller was never shown; partial then carries +// the fact without the names. +func (s *Server) cnpgWorkspaceScope(r *http.Request, namespaces []string, group, resource string) (allowed, denied []string, partial, any bool) { + if noNamespaceAccess(namespaces) { + return []string{}, nil, false, false + } + if s.canRead(r, group, resource, "", "list") { + return namespaces, nil, false, true + } + candidates := namespaces + if candidates == nil { + candidates = allNamespaceNames() + } + if len(candidates) == 0 { + return []string{}, nil, false, false + } + allowed = s.filterNamespacesByCanRead(r, group, resource, "list", candidates) + partial = len(allowed) < len(candidates) + if namespaces != nil { + for _, ns := range candidates { + if !slices.Contains(allowed, ns) { + denied = append(denied, ns) + } + } + sort.Strings(denied) + } + return allowed, denied, partial, len(allowed) > 0 +} + +func accessFromScope(allowed []string, partial bool) cnpgKindAccess { + acc := cnpgKindAccess{state: cnpgCoverageFull, all: allowed == nil} + if partial { + acc.state = cnpgCoveragePartial + } + if allowed != nil { + acc.namespaces = make(map[string]bool, len(allowed)) + for _, ns := range allowed { + acc.namespaces[ns] = true + } + } + return acc +} + +func (s *Server) cnpgWorkspaceReadKind(r *http.Request, cache *k8s.ResourceCache, k cnpgWorkspaceKind, namespaces []string) (cnpgKindAccess, []string, []*unstructured.Unstructured) { + var acc cnpgKindAccess + var denied, readNamespaces []string + if k.clusterScoped { + if !s.canRead(r, k.group, k.resource, "", "list") { + return cnpgKindAccess{state: cnpgCoverageDenied}, nil, nil + } + acc = cnpgKindAccess{state: cnpgCoverageFull, all: true} + } else { + allowed, d, partial, ok := s.cnpgWorkspaceScope(r, namespaces, k.group, k.resource) + if !ok { + return cnpgKindAccess{state: cnpgCoverageDenied}, nil, nil + } + acc, denied, readNamespaces = accessFromScope(allowed, partial), d, allowed + } + + list, err := readCNPGKind(r.Context(), cache, k, readNamespaces) + switch { + case err == nil: + return acc, denied, list + case errors.Is(err, k8s.ErrUnknownDynamicKind): + return cnpgKindAccess{state: cnpgCoverageNotInstalled}, nil, nil + case errors.Is(err, errDynamicNotSynced): + return cnpgKindAccess{state: cnpgCoverageSyncing}, nil, nil + default: + log.Printf("[cnpg] Failed to list %s.%s for workspace: %v", k.kind, k.group, err) + return cnpgKindAccess{state: cnpgCoverageError}, nil, nil + } +} + +func readCNPGKind(ctx context.Context, cache *k8s.ResourceCache, k cnpgWorkspaceKind, namespaces []string) ([]*unstructured.Unstructured, error) { + if namespaces == nil { + return filterCNPGGroup(listDynamicSynced(ctx, cache, k.kind, k.group, "")) + } + var out []*unstructured.Unstructured + for _, ns := range namespaces { + list, err := filterCNPGGroup(listDynamicSynced(ctx, cache, k.kind, k.group, ns)) + if err != nil { + return nil, err + } + out = append(out, list...) + } + return out, nil +} + +// filterCNPGGroup drops anything whose apiVersion is not a CNPG group, so a +// Velero Backup or a CAPI Cluster can never ride along on a kind-name match. +func filterCNPGGroup(items []*unstructured.Unstructured, err error) ([]*unstructured.Unstructured, error) { + if err != nil { + return nil, err + } + out := items[:0:0] + for _, u := range items { + if u == nil { + continue + } + if g := u.GroupVersionKind().Group; g != cnpgGroup && g != cnpgBarmanGroup { + continue + } + out = append(out, u) + } + return out, nil +} + +func sortCNPGObjects(items []*unstructured.Unstructured) { + sort.SliceStable(items, func(i, j int) bool { + if items[i].GetNamespace() != items[j].GetNamespace() { + return items[i].GetNamespace() < items[j].GetNamespace() + } + return items[i].GetName() < items[j].GetName() + }) +} + +func cnpgBackupTime(u *unstructured.Unstructured) time.Time { + for _, field := range []string{"stoppedAt", "startedAt"} { + if v, _, _ := unstructured.NestedString(u.Object, "status", field); v != "" { + if t, err := time.Parse(time.RFC3339, v); err == nil { + return t + } + } + } + return u.GetCreationTimestamp().Time +} + +// windowCNPGBackups keeps every in-flight Backup, settled ones from the last +// week, and each Cluster's newest completed Backup whatever its age. Sorted by +// namespace, newest first within it. +func windowCNPGBackups(items []*unstructured.Unstructured, now time.Time) ([]*unstructured.Unstructured, int) { + newestCompleted := map[string]*unstructured.Unstructured{} + for _, u := range items { + if phase, _, _ := unstructured.NestedString(u.Object, "status", "phase"); phase != "completed" { + continue + } + clusterName, _, _ := unstructured.NestedString(u.Object, "spec", "cluster", "name") + key := u.GetNamespace() + "\x00" + clusterName + if cur, ok := newestCompleted[key]; !ok || cnpgBackupTime(u).After(cnpgBackupTime(cur)) { + newestCompleted[key] = u + } + } + keepNewest := make(map[*unstructured.Unstructured]bool, len(newestCompleted)) + for _, u := range newestCompleted { + keepNewest[u] = true + } + + cutoff := now.Add(-cnpgBackupWindow) + kept := make([]*unstructured.Unstructured, 0, len(items)) + omitted := 0 + for _, u := range items { + phase, _, _ := unstructured.NestedString(u.Object, "status", "phase") + settled := phase == "completed" || phase == "failed" + if !settled || keepNewest[u] || !cnpgBackupTime(u).Before(cutoff) { + kept = append(kept, u) + continue + } + omitted++ + } + sort.SliceStable(kept, func(i, j int) bool { + if kept[i].GetNamespace() != kept[j].GetNamespace() { + return kept[i].GetNamespace() < kept[j].GetNamespace() + } + ti, tj := cnpgBackupTime(kept[i]), cnpgBackupTime(kept[j]) + if !ti.Equal(tj) { + return ti.After(tj) + } + return kept[i].GetName() < kept[j].GetName() + }) + return kept, omitted +} + +type cnpgWorkspacePodMeta struct { + Name string `json:"name"` + Namespace string `json:"namespace"` + UID types.UID `json:"uid"` + Labels map[string]string `json:"labels,omitempty"` + OwnerReferences []metav1.OwnerReference `json:"ownerReferences,omitempty"` + CreationTimestamp metav1.Time `json:"creationTimestamp"` +} + +type cnpgWorkspaceContainerStatus struct { + Name string `json:"name"` + Ready bool `json:"ready"` + RestartCount int32 `json:"restartCount"` + State corev1.ContainerState `json:"state"` +} + +type cnpgWorkspacePod struct { + APIVersion string `json:"apiVersion"` + Kind string `json:"kind"` + Metadata cnpgWorkspacePodMeta `json:"metadata"` + Spec struct { + NodeName string `json:"nodeName,omitempty"` + } `json:"spec"` + Status struct { + Phase corev1.PodPhase `json:"phase,omitempty"` + PodIP string `json:"podIP,omitempty"` + StartTime *metav1.Time `json:"startTime,omitempty"` + Conditions []corev1.PodCondition `json:"conditions,omitempty"` + ContainerStatuses []cnpgWorkspaceContainerStatus `json:"containerStatuses,omitempty"` + } `json:"status"` +} + +// isCNPGInstancePod requires the controller ownerReference to name a visible +// Cluster by UID, not just by name: a label alone is something any workload +// can carry, and a Pod left behind by a deleted Cluster must not be attributed +// to a new one created under the same name. clusterUIDs is keyed ns/name. +func isCNPGInstancePod(p *corev1.Pod, clusterUIDs map[string]types.UID) bool { + clusterName := p.Labels["cnpg.io/cluster"] + if clusterName == "" { + return false + } + uid, ok := clusterUIDs[p.Namespace+"/"+clusterName] + if !ok || uid == "" { + return false + } + for _, ref := range p.OwnerReferences { + if ref.Controller == nil || !*ref.Controller || ref.Kind != "Cluster" || ref.Name != clusterName || ref.UID != uid { + continue + } + if gv, err := schema.ParseGroupVersion(ref.APIVersion); err == nil && gv.Group == cnpgGroup { + return true + } + } + return false +} + +func cnpgClusterUIDs(clusters []*unstructured.Unstructured) map[string]types.UID { + out := make(map[string]types.UID, len(clusters)) + for _, c := range clusters { + out[c.GetNamespace()+"/"+c.GetName()] = c.GetUID() + } + return out +} + +func trimCNPGPod(p *corev1.Pod) cnpgWorkspacePod { + out := cnpgWorkspacePod{APIVersion: "v1", Kind: "Pod"} + out.Metadata = cnpgWorkspacePodMeta{ + Name: p.Name, + Namespace: p.Namespace, + UID: p.UID, + Labels: p.Labels, + OwnerReferences: p.OwnerReferences, + CreationTimestamp: p.CreationTimestamp, + } + out.Spec.NodeName = p.Spec.NodeName + out.Status.Phase = p.Status.Phase + out.Status.PodIP = p.Status.PodIP + out.Status.StartTime = p.Status.StartTime + out.Status.Conditions = p.Status.Conditions + for _, cs := range p.Status.ContainerStatuses { + out.Status.ContainerStatuses = append(out.Status.ContainerStatuses, cnpgWorkspaceContainerStatus{ + Name: cs.Name, Ready: cs.Ready, RestartCount: cs.RestartCount, State: cs.State, + }) + } + return out +} + +// cnpgTypedScope resolves where the caller may list a typed kind and which of +// those namespaces Radar's informer actually holds. The informer may itself be +// namespace-scoped when Radar's own identity cannot list the kind +// cluster-wide; what it does not hold is unread, not empty. read is nil for +// "every namespace". +func (s *Server) cnpgTypedScope(r *http.Request, cache *k8s.ResourceCache, namespaces []string, group, resource string) (acc cnpgKindAccess, denied, read []string) { + allowed, denied, partial, ok := s.cnpgWorkspaceScope(r, namespaces, group, resource) + if !ok { + return cnpgKindAccess{state: cnpgCoverageDenied}, nil, []string{} + } + within := capacityNamespacesWithinCache(cache, resource, allowed) + if within.unavailable { + log.Printf("[cnpg] %s cache does not cover the requested scope", resource) + return cnpgKindAccess{state: cnpgCoverageError}, nil, []string{} + } + if allowed != nil { + for _, ns := range allowed { + if slices.Contains(within.namespaces, ns) { + continue + } + partial = true + if namespaces != nil { + denied = append(denied, ns) + } + } + sort.Strings(denied) + } + acc = accessFromScope(within.namespaces, partial || within.partial) + return acc, denied, within.namespaces +} + +// cnpgWorkspaceReadPods returns the instance Pods of visible Clusters, plus +// the namespace/name set of what it returned — the only Pods whose issues the +// response may carry. +func (s *Server) cnpgWorkspaceReadPods(r *http.Request, cache *k8s.ResourceCache, namespaces []string, clusterUIDs map[string]types.UID) (cnpgKindAccess, []string, []any, map[string]bool) { + out := []any{} + returned := map[string]bool{} + acc, denied, read := s.cnpgTypedScope(r, cache, namespaces, "", "pods") + if acc.state == cnpgCoverageDenied || acc.state == cnpgCoverageError { + return acc, nil, out, returned + } + if cache.Pods() == nil { + log.Printf("[cnpg] Pod cache unavailable for workspace") + return cnpgKindAccess{state: cnpgCoverageError}, nil, out, returned + } + + pods := listPodsScoped(cache.Pods(), read) + sort.Slice(pods, func(i, j int) bool { + if pods[i].Namespace != pods[j].Namespace { + return pods[i].Namespace < pods[j].Namespace + } + return pods[i].Name < pods[j].Name + }) + for _, p := range pods { + if p != nil && isCNPGInstancePod(p, clusterUIDs) { + out = append(out, trimCNPGPod(p)) + returned[p.Namespace+"/"+p.Name] = true + } + } + return acc, denied, out, returned +} + +var cnpgWorkspaceKeyByGroupKind = func() map[string]string { + m := make(map[string]string, len(cnpgWorkspaceKinds)) + for _, k := range cnpgWorkspaceKinds { + m[k.group+"/"+k.kind] = k.key + } + return m +}() + +// cnpgWorkspaceIssues runs the same composition /api/issues serves, but reads +// the flat evidence rows: the grouped view folds instance-Pod evidence into +// the owning Cluster's row, which would hand Pod failure detail to a caller +// who may list Clusters but not Pods. A row is kept only when its own subject +// is visible here — a CNPG kind covered in its namespace, or an instance Pod +// this response returned. IDs are the subject-derived IDs /api/issues uses. +func (s *Server) cnpgWorkspaceIssues(r *http.Request, namespaces []string, access map[string]cnpgKindAccess, instancePods map[string]bool) []CNPGWorkspaceIssue { + out := []CNPGWorkspaceIssue{} + if noNamespaceAccess(namespaces) { + return out + } + provider := issues.NewCacheProvider() + if provider == nil { + return out + } + composed, _ := issues.ComposeWithStats(provider, issues.Filters{ + Namespaces: namespaces, + Limit: issues.NoLimit, + CanReadClusterScoped: s.issueClusterScopedAccess(r), + CanReadRelated: s.issueRelatedResourceAccess(r), + }) + for _, iss := range composed { + if !cnpgWorkspaceIssueVisible(iss, access, instancePods) { + continue + } + out = append(out, CNPGWorkspaceIssue{ + ID: iss.ID, + Severity: iss.Severity, + Category: iss.Category, + Kind: iss.Kind, + Group: iss.Group, + Namespace: iss.Namespace, + Name: iss.Name, + Reason: iss.Reason, + Message: iss.Message, + Cause: iss.Cause, + Action: iss.Action, + FirstSeen: iss.FirstSeen, + }) + } + return out +} + +func cnpgWorkspaceIssueVisible(iss issues.Issue, access map[string]cnpgKindAccess, instancePods map[string]bool) bool { + if iss.Group == "" && iss.Kind == "Pod" { + return access[cnpgWorkspacePodsKey].covers(iss.Namespace) && instancePods[iss.Namespace+"/"+iss.Name] + } + if iss.Group != cnpgGroup && iss.Group != cnpgBarmanGroup { + return false + } + key, ok := cnpgWorkspaceKeyByGroupKind[iss.Group+"/"+iss.Kind] + return ok && access[key].covers(iss.Namespace) +} + +// cnpgWorkspaceAudit reports the declarative-backup posture finding only for +// Clusters whose namespace had its ScheduledBackups read: without that list, +// "no schedule targets this cluster" is an absence nobody established. +func cnpgWorkspaceAudit(clusters, scheduled []*unstructured.Unstructured, schedAccess cnpgKindAccess) []CNPGWorkspaceAuditFinding { + out := []CNPGWorkspaceAuditFinding{} + var subjects []*unstructured.Unstructured + for _, c := range clusters { + if schedAccess.covers(c.GetNamespace()) { + subjects = append(subjects, c) + } + } + if len(subjects) == 0 { + return out + } + results := bp.RunChecks(&bp.CheckInput{ + CNPGClusters: subjects, + CNPGScheduledBackups: scheduled, + CNPGScheduledBackupsAuthoritative: true, + }) + results = applyAuditSettings(results, getAuditConfig()) + if results == nil { + return out + } + for _, f := range results.Findings { + if f.CheckID != cnpgNoDeclarativeBackupCheckID { + continue + } + out = append(out, CNPGWorkspaceAuditFinding{ + CheckID: f.CheckID, + Severity: f.Severity, + Kind: f.Kind, + Group: f.Group, + Namespace: f.Namespace, + Name: f.Name, + Message: f.Message, + }) + } + sort.SliceStable(out, func(i, j int) bool { + if out[i].Namespace != out[j].Namespace { + return out[i].Namespace < out[j].Namespace + } + return out[i].Name < out[j].Name + }) + return out +} diff --git a/internal/server/cnpg_workspace_test.go b/internal/server/cnpg_workspace_test.go new file mode 100644 index 0000000000..02152adb16 --- /dev/null +++ b/internal/server/cnpg_workspace_test.go @@ -0,0 +1,584 @@ +package server + +import ( + "context" + "encoding/json" + "net/http" + "strings" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/apimachinery/pkg/types" + dynamicfake "k8s.io/client-go/dynamic/fake" + + "github.com/skyhook-io/radar/internal/auth" + "github.com/skyhook-io/radar/internal/k8s" +) + +func cnpgTestResource(group, kind, resource string, namespaced bool) k8s.APIResource { + return k8s.APIResource{Group: group, Version: "v1", Kind: kind, Name: resource, Namespaced: namespaced, IsCRD: true, Verbs: []string{"get", "list", "watch"}} +} + +var cnpgWorkspaceTestKinds = func() []k8s.APIResource { + var out []k8s.APIResource + for _, k := range cnpgWorkspaceKinds { + out = append(out, cnpgTestResource(k.group, k.kind, k.resource, !k.clusterScoped)) + } + out = append(out, + cnpgTestResource(veleroGroup, "Backup", "backups", true), + k8s.APIResource{Group: "cluster.x-k8s.io", Version: "v1beta1", Kind: "Cluster", Name: "clusters", Namespaced: true, IsCRD: true, Verbs: []string{"get", "list", "watch"}}, + ) + return out +}() + +func seedCNPGWorkspace(t *testing.T, kinds []k8s.APIResource, objs ...runtime.Object) { + t.Helper() + listKinds := map[schema.GroupVersionResource]string{} + for _, k := range kinds { + listKinds[schema.GroupVersionResource{Group: k.Group, Version: k.Version, Resource: k.Name}] = k.Kind + "List" + } + dyn := dynamicfake.NewSimpleDynamicClientWithCustomListKinds(runtime.NewScheme(), listKinds, objs...) + if err := k8s.InitTestDynamicResourceCache(dyn, kinds); err != nil { + t.Fatalf("seed cnpg: %v", err) + } + t.Cleanup(k8s.ResetTestDynamicState) +} + +func cnpgObj(apiVersion, kind, ns, name string, spec, status map[string]any) *unstructured.Unstructured { + meta := map[string]any{"name": name, "creationTimestamp": time.Now().Add(-time.Hour).UTC().Format(time.RFC3339)} + if ns != "" { + meta["namespace"] = ns + } + obj := map[string]any{"apiVersion": apiVersion, "kind": kind, "metadata": meta} + if spec != nil { + obj["spec"] = spec + } + if status != nil { + obj["status"] = status + } + return &unstructured.Unstructured{Object: obj} +} + +func cnpgBackup(ns, name, cluster, phase string, stoppedAt time.Time) *unstructured.Unstructured { + status := map[string]any{"phase": phase} + if !stoppedAt.IsZero() { + status["stoppedAt"] = stoppedAt.UTC().Format(time.RFC3339) + } + return cnpgObj("postgresql.cnpg.io/v1", "Backup", ns, name, map[string]any{"cluster": map[string]any{"name": cluster}}, status) +} + +func decodeWorkspace(t *testing.T, resp *http.Response) CNPGWorkspaceResponse { + t.Helper() + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + t.Fatalf("status = %d, want 200", resp.StatusCode) + } + var out CNPGWorkspaceResponse + if err := json.NewDecoder(resp.Body).Decode(&out); err != nil { + t.Fatalf("decode: %v", err) + } + return out +} + +func getWorkspaceNoAuth(t *testing.T, query string) CNPGWorkspaceResponse { + t.Helper() + resp, err := http.Get(testServer.URL + "/api/cnpg/workspace" + query) + if err != nil { + t.Fatalf("GET: %v", err) + } + return decodeWorkspace(t, resp) +} + +func objectNames(objs []any) []string { + var out []string + for _, o := range objs { + m, _ := o.(map[string]any) + meta, _ := m["metadata"].(map[string]any) + name, _ := meta["name"].(string) + out = append(out, name) + } + return out +} + +func containsName(objs []any, name string) bool { + for _, n := range objectNames(objs) { + if n == name { + return true + } + } + return false +} + +func assertEveryKey(t *testing.T, got CNPGWorkspaceResponse) { + t.Helper() + keys := []string{cnpgWorkspacePodsKey} + for _, k := range cnpgWorkspaceKinds { + keys = append(keys, k.key) + } + for _, k := range keys { + if _, ok := got.Coverage[k]; !ok { + t.Errorf("coverage missing key %q", k) + } + if objs, ok := got.Objects[k]; !ok || objs == nil { + t.Errorf("objects missing key %q (or null)", k) + } + } +} + +func TestCNPGWorkspace_NotInstalled(t *testing.T) { + seedCNPGWorkspace(t, []k8s.APIResource{cnpgTestResource(veleroGroup, "Backup", "backups", true)}) + got := getWorkspaceNoAuth(t, "") + if got.Installed { + t.Error("installed = true on a cluster without CloudNativePG") + } + assertEveryKey(t, got) + for k, c := range got.Coverage { + if c.State != cnpgCoverageNotInstalled { + t.Errorf("coverage[%s] = %q, want notInstalled", k, c.State) + } + } +} + +func seedCNPGPods(t *testing.T, pods ...*corev1.Pod) { + t.Helper() + ctx := context.Background() + for _, p := range pods { + if _, err := testFakeClient.CoreV1().Pods(p.Namespace).Create(ctx, p, metav1.CreateOptions{}); err != nil { + t.Fatalf("create pod %s: %v", p.Name, err) + } + t.Cleanup(func() { + _ = testFakeClient.CoreV1().Pods(p.Namespace).Delete(context.Background(), p.Name, metav1.DeleteOptions{}) + }) + } + lister := k8s.GetResourceCache().Pods() + deadline := time.Now().Add(5 * time.Second) + for { + seen := 0 + for _, p := range pods { + if _, err := lister.Pods(p.Namespace).Get(p.Name); err == nil { + seen++ + } + } + if seen == len(pods) { + return + } + if time.Now().After(deadline) { + t.Fatalf("pods did not reach the cache") + } + time.Sleep(20 * time.Millisecond) + } +} + +func cnpgPod(ns, name, clusterLabel string, owners ...metav1.OwnerReference) *corev1.Pod { + return &corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{ + Name: name, Namespace: ns, + Labels: map[string]string{"cnpg.io/cluster": clusterLabel, "cnpg.io/instanceRole": "primary"}, + OwnerReferences: owners, + }, + Spec: corev1.PodSpec{NodeName: "node-1", Containers: []corev1.Container{{Name: "postgres", Image: "pg:17"}}}, + Status: corev1.PodStatus{ + Phase: corev1.PodRunning, + PodIP: "10.0.0.5", + ContainerStatuses: []corev1.ContainerStatus{{ + Name: "postgres", Ready: true, RestartCount: 2, Image: "pg:17", + State: corev1.ContainerState{Running: &corev1.ContainerStateRunning{}}, + }}, + }, + } +} + +func TestCNPGWorkspace_AuthDisabledReturnsEverythingAndOnlyOwnedInstancePods(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + withUID(cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgws", "pg-orders", map[string]any{"instances": int64(1)}, nil), "c-uid"), + cnpgObj("postgresql.cnpg.io/v1", "Pooler", "pgws", "pg-orders-rw", map[string]any{"cluster": map[string]any{"name": "pg-orders"}}, nil), + cnpgObj("postgresql.cnpg.io/v1", "ClusterImageCatalog", "", "pg-fleet", nil, nil), + cnpgObj("barmancloud.cnpg.io/v1", "ObjectStore", "pgws", "store", nil, nil), + ) + owner := metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "c-uid", Controller: boolPtr(true)} + seedCNPGPods(t, + cnpgPod("pgws", "pg-orders-1", "pg-orders", owner), + cnpgPod("pgws", "impostor-1", "pg-orders"), + cnpgPod("pgws", "capi-owned-1", "pg-orders", metav1.OwnerReference{APIVersion: "cluster.x-k8s.io/v1beta1", Kind: "Cluster", Name: "pg-orders", UID: "x"}), + cnpgPod("pgws", "other-owner-1", "pg-orders", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-billing", UID: "y"}), + cnpgPod("pgws", "stale-uid-1", "pg-orders", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "deleted-cluster-uid", Controller: boolPtr(true)}), + cnpgPod("pgws", "not-controller-1", "pg-orders", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "c-uid"}), + ) + + got := getWorkspaceNoAuth(t, "") + if !got.Installed { + t.Fatal("installed = false with CNPG kinds discovered") + } + assertEveryKey(t, got) + for k, c := range got.Coverage { + if c.State != cnpgCoverageFull { + t.Errorf("coverage[%s] = %+v, want full with auth disabled", k, c) + } + } + if got.Namespaces != nil { + t.Errorf("namespaces = %v, want null for an unfiltered view", got.Namespaces) + } + for key, name := range map[string]string{"clusters": "pg-orders", "poolers": "pg-orders-rw", "clusterImageCatalogs": "pg-fleet", "objectStores": "store"} { + if !containsName(got.Objects[key], name) { + t.Errorf("objects[%s] = %v, want %s", key, objectNames(got.Objects[key]), name) + } + } + pods := objectNames(got.Objects["pods"]) + if len(pods) != 1 || pods[0] != "pg-orders-1" { + t.Fatalf("pods = %v, want only the CNPG-owned instance pod", pods) + } + pod := got.Objects["pods"][0].(map[string]any) + if _, ok := pod["spec"].(map[string]any)["containers"]; ok { + t.Error("pod spec was not trimmed") + } + cs := pod["status"].(map[string]any)["containerStatuses"].([]any)[0].(map[string]any) + if cs["restartCount"].(float64) != 2 || cs["ready"] != true { + t.Errorf("container status = %v", cs) + } + if _, ok := cs["image"]; ok { + t.Error("container status carried fields beyond the trimmed set") + } +} + +func TestCNPGWorkspace_CollidingKindsNeverLeak(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgws", "pg-orders", nil, nil), + cnpgBackup("pgws", "pg-orders-b1", "pg-orders", "running", time.Time{}), + cnpgObj("velero.io/v1", "Backup", "pgws", "velero-nightly", nil, map[string]any{"phase": "Completed"}), + cnpgObj("cluster.x-k8s.io/v1beta1", "Cluster", "pgws", "capi-workload", nil, nil), + ) + got := getWorkspaceNoAuth(t, "") + if containsName(got.Objects["backups"], "velero-nightly") { + t.Error("a Velero Backup was returned as a CNPG Backup") + } + if containsName(got.Objects["clusters"], "capi-workload") { + t.Error("a CAPI Cluster was returned as a CNPG Cluster") + } + if !containsName(got.Objects["backups"], "pg-orders-b1") || !containsName(got.Objects["clusters"], "pg-orders") { + t.Errorf("CNPG objects missing: backups=%v clusters=%v", objectNames(got.Objects["backups"]), objectNames(got.Objects["clusters"])) + } +} + +func TestCNPGWorkspace_DeniedKindAndItsIssuesAreWithheld(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pg", "pg-orders", nil, nil), + cnpgBackup("pg", "pg-orders-broken", "pg-orders", "failed", time.Now().Add(-time.Hour)), + ) + // Warm the Backup informer so the issues engine can see the failed Backup + // whichever caller asks; otherwise a denied answer would pass vacuously. + if _, err := listDynamicSynced(context.Background(), k8s.GetResourceCache(), "Backup", cnpgGroup, ""); err != nil { + t.Fatalf("warm backups: %v", err) + } + + env := newAuthTestServer(t) + for _, u := range []struct { + name string + backupsListed bool + }{{"reader", true}, {"no-backups", false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pg"}} + allow(perms, cnpgGroup, "clusters", "", true) + allow(perms, cnpgGroup, "backups", "", u.backupsListed) + allow(perms, cnpgGroup, "backups", "pg", u.backupsListed) + env.srv.permCache.Set(u.name, nil, perms) + } + + hasBackupIssue := func(got CNPGWorkspaceResponse) bool { + for _, iss := range got.Issues { + if iss.Kind == "Backup" && iss.Name == "pg-orders-broken" { + return true + } + } + return false + } + + control := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "reader", "")) + if control.Coverage["backups"].State != cnpgCoverageFull || !containsName(control.Objects["backups"], "pg-orders-broken") { + t.Fatalf("control: backups coverage=%+v objects=%v", control.Coverage["backups"], objectNames(control.Objects["backups"])) + } + if !hasBackupIssue(control) { + t.Fatalf("control: the failed Backup raised no issue, so the denied case would prove nothing: %+v", control.Issues) + } + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "no-backups", "")) + if got.Coverage["backups"].State != cnpgCoverageDenied { + t.Errorf("backups coverage = %+v, want denied", got.Coverage["backups"]) + } + if len(got.Objects["backups"]) != 0 { + t.Errorf("backups = %v, want [] when denied", objectNames(got.Objects["backups"])) + } + if hasBackupIssue(got) { + t.Error("an issue on a Backup the caller cannot list was returned") + } + if got.Coverage["clusters"].State != cnpgCoverageFull || !containsName(got.Objects["clusters"], "pg-orders") { + t.Errorf("clusters coverage=%+v objects=%v, want full", got.Coverage["clusters"], objectNames(got.Objects["clusters"])) + } +} + +func TestCNPGWorkspace_PartialNamespaceCoverage(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "a", "pg-a", nil, nil), + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "b", "pg-b", nil, nil), + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "c", "pg-c", nil, nil), + ) + env := newAuthTestServer(t) + perms := &auth.UserPermissions{AllowedNamespaces: []string{"a", "b"}} + allow(perms, cnpgGroup, "clusters", "", false) + allow(perms, cnpgGroup, "clusters", "a", true) + allow(perms, cnpgGroup, "clusters", "b", false) + env.srv.permCache.Set("scoped", nil, perms) + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "scoped", "")) + cov := got.Coverage["clusters"] + if cov.State != cnpgCoveragePartial || len(cov.DeniedNamespaces) != 1 || cov.DeniedNamespaces[0] != "b" { + t.Errorf("clusters coverage = %+v, want partial denied [b]", cov) + } + if len(cov.AllowedNamespaces) != 1 || cov.AllowedNamespaces[0] != "a" { + t.Errorf("allowedNamespaces = %v, want [a]", cov.AllowedNamespaces) + } + names := objectNames(got.Objects["clusters"]) + if len(names) != 1 || names[0] != "pg-a" { + t.Errorf("clusters = %v, want only pg-a", names) + } + + // A view filter narrows the scope; the denied list never grows past it. + filtered := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace?namespaces=a", "scoped", "")) + if filtered.Coverage["clusters"].State != cnpgCoverageFull { + t.Errorf("filtered to a: coverage = %+v, want full", filtered.Coverage["clusters"]) + } + if len(filtered.Namespaces) != 1 || filtered.Namespaces[0] != "a" { + t.Errorf("namespaces = %v, want [a]", filtered.Namespaces) + } +} + +func TestCNPGWorkspace_ClusterImageCatalogNeedsClusterScopeGrant(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "ClusterImageCatalog", "", "pg-fleet", nil, nil), + ) + env := newAuthTestServer(t) + nsOnly := &auth.UserPermissions{AllowedNamespaces: []string{"pg"}} + nsOnly.SetCanI("list", cnpgGroup, "clusterimagecatalogs", "pg", true) + allow(nsOnly, cnpgGroup, "clusterimagecatalogs", "", false) + env.srv.permCache.Set("ns-only", nil, nsOnly) + + clusterWide := &auth.UserPermissions{AllowedNamespaces: []string{"pg"}} + allow(clusterWide, cnpgGroup, "clusterimagecatalogs", "", true) + env.srv.permCache.Set("cluster-wide", nil, clusterWide) + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "ns-only", "")) + if got.Coverage["clusterImageCatalogs"].State != cnpgCoverageDenied || len(got.Objects["clusterImageCatalogs"]) != 0 { + t.Errorf("namespace-level grant exposed ClusterImageCatalogs: %+v %v", got.Coverage["clusterImageCatalogs"], objectNames(got.Objects["clusterImageCatalogs"])) + } + + got = decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace?namespaces=pg", "cluster-wide", "")) + if got.Coverage["clusterImageCatalogs"].State != cnpgCoverageFull || !containsName(got.Objects["clusterImageCatalogs"], "pg-fleet") { + t.Errorf("cluster-scope grant: %+v %v, want full with pg-fleet regardless of the view filter", got.Coverage["clusterImageCatalogs"], objectNames(got.Objects["clusterImageCatalogs"])) + } +} + +func TestWindowCNPGBackups(t *testing.T) { + now := time.Date(2026, 9, 29, 12, 0, 0, 0, time.UTC) + day := 24 * time.Hour + items := []*unstructured.Unstructured{ + cnpgBackup("pg", "running-old", "orders", "running", time.Time{}), + cnpgBackup("pg", "orders-recent", "orders", "completed", now.Add(-2*day)), + cnpgBackup("pg", "orders-old", "orders", "completed", now.Add(-20*day)), + cnpgBackup("pg", "orders-failed-old", "orders", "failed", now.Add(-30*day)), + cnpgBackup("pg", "orders-failed-new", "orders", "failed", now.Add(-1*day)), + cnpgBackup("pg", "billing-only-old", "billing", "completed", now.Add(-40*day)), + cnpgBackup("pg", "billing-older", "billing", "completed", now.Add(-50*day)), + cnpgBackup("aa", "other-ns", "orders", "completed", now.Add(-60*day)), + } + // A running Backup older than the window stays: in flight is never settled. + items[0].Object["metadata"].(map[string]any)["creationTimestamp"] = now.Add(-90 * day).Format(time.RFC3339) + + kept, omitted := windowCNPGBackups(items, now) + var names []string + for _, u := range kept { + names = append(names, u.GetName()) + } + want := []string{"other-ns", "orders-failed-new", "orders-recent", "billing-only-old", "running-old"} + if len(names) != len(want) { + t.Fatalf("kept = %v, want %v", names, want) + } + for i := range want { + if names[i] != want[i] { + t.Fatalf("kept = %v, want %v (namespace, then newest first)", names, want) + } + } + if omitted != 3 { + t.Errorf("omitted = %d, want 3 (orders-old, orders-failed-old, billing-older)", omitted) + } +} + +func TestCNPGWorkspace_BackupsOmittedIsReported(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgBackup("pgws", "new", "orders", "completed", time.Now().Add(-time.Hour)), + cnpgBackup("pgws", "ancient", "orders", "completed", time.Now().Add(-30*24*time.Hour)), + ) + got := getWorkspaceNoAuth(t, "") + if got.BackupsOmitted != 1 || containsName(got.Objects["backups"], "ancient") { + t.Errorf("backupsOmitted=%d backups=%v, want 1 and ancient omitted", got.BackupsOmitted, objectNames(got.Objects["backups"])) + } +} + +func TestCNPGWorkspace_AuditNeedsScheduledBackupEvidence(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pg", "unscheduled", nil, nil), + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pg", "scheduled", nil, nil), + cnpgObj("postgresql.cnpg.io/v1", "ScheduledBackup", "pg", "nightly", map[string]any{"cluster": map[string]any{"name": "scheduled"}}, nil), + ) + env := newAuthTestServer(t) + for _, u := range []struct { + name string + sched bool + }{{"sees-schedules", true}, {"no-schedules", false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pg"}} + allow(perms, cnpgGroup, "clusters", "", true) + allow(perms, cnpgGroup, "scheduledbackups", "", u.sched) + allow(perms, cnpgGroup, "scheduledbackups", "pg", u.sched) + env.srv.permCache.Set(u.name, nil, perms) + } + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "sees-schedules", "")) + if len(got.Audit) != 1 || got.Audit[0].Name != "unscheduled" || got.Audit[0].CheckID != cnpgNoDeclarativeBackupCheckID { + t.Errorf("audit = %+v, want one %s finding on unscheduled", got.Audit, cnpgNoDeclarativeBackupCheckID) + } + + got = decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "no-schedules", "")) + if got.Coverage["scheduledBackups"].State != cnpgCoverageDenied { + t.Errorf("scheduledBackups coverage = %+v, want denied", got.Coverage["scheduledBackups"]) + } + if len(got.Audit) != 0 { + t.Errorf("audit = %+v, want none — without the ScheduledBackup list the absence is unestablished", got.Audit) + } +} + +func withUID(u *unstructured.Unstructured, uid string) *unstructured.Unstructured { + u.SetUID(types.UID(uid)) + return u +} + +func TestIsCNPGInstancePod(t *testing.T) { + uids := map[string]types.UID{"pg/x": "x-uid"} + ref := func(uid string, controller bool) metav1.OwnerReference { + return metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "x", UID: types.UID(uid), Controller: boolPtr(controller)} + } + for _, c := range []struct { + name string + pod *corev1.Pod + uids map[string]types.UID + want bool + }{ + {"controller ref to the visible Cluster", cnpgPod("pg", "x-1", "x", ref("x-uid", true)), uids, true}, + {"left behind by a deleted Cluster of the same name", cnpgPod("pg", "x-1", "x", ref("old-uid", true)), uids, false}, + {"non-controller owner", cnpgPod("pg", "x-1", "x", ref("x-uid", false)), uids, false}, + {"Cluster not visible", cnpgPod("pg", "x-1", "x", ref("x-uid", true)), map[string]types.UID{}, false}, + {"Cluster of that name in another namespace", cnpgPod("other", "x-1", "x", ref("x-uid", true)), uids, false}, + {"no cluster label", func() *corev1.Pod { p := cnpgPod("pg", "x-1", "x", ref("x-uid", true)); p.Labels = nil; return p }(), uids, false}, + } { + t.Run(c.name, func(t *testing.T) { + if got := isCNPGInstancePod(c.pod, c.uids); got != c.want { + t.Errorf("isCNPGInstancePod = %v, want %v", got, c.want) + } + }) + } +} + +// The grouped issue view folds instance-Pod evidence into the owning Cluster's +// row. A caller who may list Clusters but not Pods must not receive that +// evidence in any form; one who may list Pods receives it on the Pod itself. +func TestCNPGWorkspace_PodEvidenceFollowsPodAccess(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + withUID(cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgev", "pg-orders", map[string]any{"instances": int64(1)}, nil), "orders-uid"), + ) + crashing := cnpgPod("pgev", "pg-orders-1", "pg-orders", metav1.OwnerReference{ + APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "orders-uid", Controller: boolPtr(true), + }) + crashing.Status.ContainerStatuses[0] = corev1.ContainerStatus{ + Name: "postgres", Ready: false, RestartCount: 9, Image: "pg:17", + State: corev1.ContainerState{Waiting: &corev1.ContainerStateWaiting{Reason: "CrashLoopBackOff", Message: "back-off restarting failed container"}}, + } + seedCNPGPods(t, crashing) + + env := newAuthTestServer(t) + for _, u := range []struct { + name string + pods bool + }{{"with-pods", true}, {"clusters-only", false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pgev"}} + allow(perms, cnpgGroup, "clusters", "", true) + allow(perms, "", "pods", "", u.pods) + allow(perms, "", "pods", "pgev", u.pods) + env.srv.permCache.Set(u.name, nil, perms) + } + + control := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "with-pods", "")) + var podIssue *CNPGWorkspaceIssue + for i, iss := range control.Issues { + if iss.Kind == "Pod" && iss.Name == "pg-orders-1" { + podIssue = &control.Issues[i] + } + } + if podIssue == nil { + t.Fatalf("with pod access: no Pod issue for the crashlooping instance, got %+v", control.Issues) + } + if podIssue.ID == "" { + t.Error("Pod issue carries no ID") + } + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "clusters-only", "")) + if got.Coverage["pods"].State != cnpgCoverageDenied || len(got.Objects["pods"]) != 0 { + t.Errorf("pods coverage=%+v objects=%v, want denied and []", got.Coverage["pods"], objectNames(got.Objects["pods"])) + } + for _, iss := range got.Issues { + if iss.Kind == "Pod" || strings.Contains(iss.Message, "CrashLoopBackOff") || strings.Contains(iss.Message, "back-off") || iss.ID == podIssue.ID { + t.Errorf("Pod evidence reached a caller without Pod access: %+v", iss) + } + } +} + +// Denied namespaces are named only when the caller supplied the candidate set. +// For a caller whose scope is "all", the candidates are every namespace Radar +// holds, and listing the denied ones would disclose them. +func TestCNPGWorkspace_DeniedNamespacesNeverComeFromTheServerInventory(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "default", "pg-default", nil, nil), + ) + env := newAuthTestServer(t) + perms := &auth.UserPermissions{} + allow(perms, cnpgGroup, "clusters", "", false) + for _, ns := range allNamespaceNames() { + allow(perms, cnpgGroup, "clusters", ns, ns == "default") + } + allow(perms, cnpgGroup, "clusters", "broken", false) + env.srv.permCache.Set("wide", nil, perms) + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "wide", "")) + cov := got.Coverage["clusters"] + if cov.State != cnpgCoveragePartial { + t.Errorf("unfiltered: clusters coverage = %+v, want partial", cov) + } + if len(cov.DeniedNamespaces) != 0 { + t.Errorf("unfiltered: deniedNamespaces = %v, want omitted — they came from Radar's namespace inventory", cov.DeniedNamespaces) + } + if len(cov.AllowedNamespaces) != 1 || cov.AllowedNamespaces[0] != "default" { + t.Errorf("unfiltered: allowedNamespaces = %v, want [default] — without it the reader cannot tell which namespaces were read", cov.AllowedNamespaces) + } + if !containsName(got.Objects["clusters"], "pg-default") { + t.Errorf("unfiltered: clusters = %v, want pg-default", objectNames(got.Objects["clusters"])) + } + + got = decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace?namespaces=default,broken", "wide", "")) + cov = got.Coverage["clusters"] + if cov.State != cnpgCoveragePartial || len(cov.DeniedNamespaces) != 1 || cov.DeniedNamespaces[0] != "broken" { + t.Errorf("filtered: clusters coverage = %+v, want partial naming broken", cov) + } + if len(cov.AllowedNamespaces) != 1 || cov.AllowedNamespaces[0] != "default" { + t.Errorf("filtered: allowedNamespaces = %v, want [default]", cov.AllowedNamespaces) + } +} diff --git a/internal/server/server.go b/internal/server/server.go index dd1a3da0ad..dbfee29052 100644 --- a/internal/server/server.go +++ b/internal/server/server.go @@ -528,6 +528,7 @@ func (s *Server) setupAppRoutes(r chi.Router) { // over a slow cluster link legitimately takes longer than that. r.Post("/pods/{namespace}/{name}/files/save", s.handlePodFileSave) r.Get("/workloads/{kind}/{namespace}/{name}/logs/stream", s.handleWorkloadLogsStream) + r.Get("/cnpg/clusters/{namespace}/{name}/logs/stream", s.handleCNPGClusterLogsStream) // AI investigation event stream via SSE — long-lived; lives outside the // 60s timeout group. The run keeps going server-side after disconnect. r.Get("/diagnose/runs/{id}/stream", s.handleDiagnoseRunStream) @@ -599,8 +600,12 @@ func (s *Server) setupAppRoutes(r chi.Router) { r.Get("/rbac/role/{kind}/{namespace}/{name}", s.handleRBACRole) r.Get("/rbac/namespace/{namespace}", s.handleRBACNamespace) r.Get("/rbac/whoami", s.handleRBACWhoami) + r.Get("/cnpg/workspace", s.handleCNPGWorkspace) + r.Get("/cnpg/operator", s.handleCNPGOperator) r.Get("/cnpg/imagecatalogs/{namespace}/{name}/clusters", s.handleCNPGCatalogUsers) r.Get("/cnpg/clusterimagecatalogs/{name}/clusters", s.handleCNPGCatalogUsers) + r.Get("/cnpg/clusters/{namespace}/{name}/logs", s.handleCNPGClusterLogs) + r.Get("/cnpg/clusters/{namespace}/{name}/activity", s.handleCNPGClusterActivity) r.Get("/velero/backupstoragelocations/{namespace}/{name}/backups", s.handleVeleroStoredBackups) // POST: creates a DownloadRequest, which is the only supported way to // read the messages behind a run's error and warning counts. diff --git a/internal/server/workload_logs.go b/internal/server/workload_logs.go index d35b221678..8f9709ad56 100644 --- a/internal/server/workload_logs.go +++ b/internal/server/workload_logs.go @@ -68,6 +68,10 @@ type workloadLogEntry struct { Timestamp string `json:"timestamp"` Content string `json:"content"` SourceLabel string `json:"sourceLabel,omitempty"` + // Parsed from a structured line by sources that know their log format. + Level string `json:"level,omitempty"` + Logger string `json:"logger,omitempty"` + Message string `json:"message,omitempty"` } type workloadLogMetadata struct { @@ -363,7 +367,7 @@ func (s *Server) authorizeWorkloadLogRead(w http.ResponseWriter, r *http.Request func (s *Server) authorizePodLogRead(w http.ResponseWriter, r *http.Request, namespace string) bool { if !s.canReadSubresource(r, "", "pods", "log", namespace, "get") { - s.writeError(w, http.StatusForbidden, "no access to pod logs in namespace "+namespace) + s.writeError(w, http.StatusForbidden, "no access to pod logs in namespace "+namespace+": requires get pods/log") return false } return true diff --git a/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx new file mode 100644 index 0000000000..0902a147d7 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx @@ -0,0 +1,223 @@ +import { formatDuration } from '../resources/resource-utils' +import { + CNPG_BARMAN_OBJECTSTORE_GROUP, + CNPG_GROUP, + getCNPGBackupStatus, + getCNPGScheduledBackupNextSchedule, + getCNPGScheduledBackupStatus, +} from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' +import { ClusterLink, NotReported, Note, ObjectProblems, PhaseBadge, SummaryShell, TimeAgo } from './CNPGSharedSummary' +import { + backupDestination, + backupsForScheduledBackup, + clustersIn, + isBackupFromSchedule, + refOf, + relationUnavailable, + scheduledBackupOf, + workspaceList, +} from './relations' + +interface SummaryProps { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +} + +const RECENT_RUNS = 5 + +function methodText(resource: any): string | null { + const m = resource?.status?.method || resource?.spec?.method + if (!m) return null + switch (m) { + case 'plugin': + return resource?.spec?.pluginConfiguration?.name ? `Plugin · ${resource.spec.pluginConfiguration.name}` : 'Plugin' + case 'barmanObjectStore': + return 'Barman object store (in-tree)' + case 'volumeSnapshot': + return 'Volume snapshot' + default: + return m + } +} + +function Timing({ resource }: { resource: any }) { + const started = resource?.status?.startedAt + const stopped = resource?.status?.stoppedAt + const startMs = started ? Date.parse(started) : NaN + const stopMs = stopped ? Date.parse(stopped) : NaN + const tookMs = Number.isFinite(startMs) && Number.isFinite(stopMs) && stopMs >= startMs ? stopMs - startMs : null + return ( + <> + + + + + {stopped ? ( + + + {tookMs !== null && · took {formatDuration(tookMs, true)}} + + ) : Number.isFinite(startMs) ? ( + Not stopped · running for {formatDuration(Date.now() - startMs, true)} + ) : ( + + )} + + + ) +} + +export function CNPGBackupSummary({ resource, workspace, onNavigate }: SummaryProps) { + const ns = resource?.metadata?.namespace ?? '' + const status = getCNPGBackupStatus(resource) + const pod = resource?.status?.instanceID?.podName + const error = resource?.status?.error + const trigger = scheduledBackupOf(resource) + const liveSchedule = trigger + ? workspaceList(workspace, 'scheduledBackups').find((s) => s?.metadata?.namespace === ns && s?.metadata?.name === trigger) + : undefined + const scheduleReplaced = !!liveSchedule && !isBackupFromSchedule(resource, liveSchedule) + const dest = backupDestination(resource, clustersIn(workspace)) + const method = methodText(resource) + + return ( + + + + Outcome + + + + + + + {pod ? : } + + + {resource?.status?.backupId ? {resource.status.backupId} : } + + + {resource?.spec?.target ? resource.spec.target : Not set · the Cluster's backup target applies} + + {error && ( + + {error} + + )} + + + Relationships + + + + + + {trigger ? ( + + ScheduledBackup{' '} + {scheduleReplaced ? ( + <> + {trigger} + · an earlier schedule of that name; the current one is a different object + + ) : ( + + )} + + ) : ( + 'On demand' + )} + + + {dest.type === 'objectStore' ? ( + + ObjectStore{' '} + + {dest.inferred && · from the Cluster's current configuration} + + ) : dest.type === 'path' ? ( + {dest.path} + ) : dest.type === 'volumeSnapshot' ? ( + 'Volume snapshot' + ) : ( + + )} + + {method ?? } + + + ) +} + +export function CNPGScheduledBackupSummary({ resource, workspace, onNavigate }: SummaryProps) { + const ns = resource?.metadata?.namespace ?? '' + const cron = resource?.spec?.schedule + const next = getCNPGScheduledBackupNextSchedule(resource) + const runsUnavailable = relationUnavailable(workspace, 'backups', ns, 'Backups') + const runs = runsUnavailable ? [] : backupsForScheduledBackup(resource, workspaceList(workspace, 'backups')) + const shown = runs.slice(0, RECENT_RUNS) + + return ( + + + + Schedule + + + + + + + + + {cron ? ( + + {cron} + · six fields, seconds first + + ) : ( + + )} + + + + + {next === '-' ? : next} + {methodText(resource) ?? 'Barman object store (in-tree) · default'} + + + shown.length ? `${shown.length} of ${runs.length}` : undefined}>Recent runs + {runsUnavailable ? ( +
+ +
+ ) : shown.length === 0 ? ( +
No Backups from this schedule are visible
+ ) : ( + + {shown.map((b) => ( + } + > + + + {b.status?.startedAt ? ( + + started + + ) : ( + + )} + + + ))} + + )} + {!runsUnavailable && (workspace?.backupsOmitted ?? 0) > 0 && Backups older than 7 days are not listed.} +
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx new file mode 100644 index 0000000000..a35db6a9bb --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx @@ -0,0 +1,222 @@ +import type { ReactNode } from 'react' +import { clsx } from 'clsx' +import { Badge } from '../ui/Badge' +import { Tooltip } from '../ui/Tooltip' +import { CNPG_BARMAN_OBJECTSTORE_GROUP, CNPG_GROUP } from '../resources/resource-utils-cnpg' +import type { CNPGFleetRow, CNPGInstance } from './workspace' +import { + FactGrid, + FactRow, + FactSource, + FactValue, + ProblemCallout, + RefLink, + SummaryHeading, + ToneDot, + toneTextClass, + CNPG_PRIMARY_BUTTON, + CNPG_SECONDARY_BUTTON, + type CNPGNavigate, +} from './primitives' + +export interface CNPGSummaryAction { + label: string + onClick: () => void + primary?: boolean +} + +function InstancePill({ pod, namespace, onNavigate }: { pod: CNPGInstance; namespace: string; onNavigate?: CNPGNavigate }) { + const tone = pod.ready === true ? 'healthy' : pod.ready === false ? 'unhealthy' : 'unknown' + const role = pod.role === 'primary' ? 'Primary' : pod.role === 'replica' ? 'Replica' : 'Role unknown' + const readiness = pod.ready === true ? 'Ready' : pod.ready === false ? 'Not ready' : 'Readiness unknown' + return ( + + + + ) +} + +export function CNPGClusterSummary({ + row, + onNavigate, + actions, + problemsLink, + extra, +}: { + row: CNPGFleetRow + onNavigate?: CNPGNavigate + actions?: CNPGSummaryAction[] + /** Link to the complete list of this cluster's findings, shown when more than one exists. */ + problemsLink?: (count: number) => ReactNode + extra?: ReactNode +}) { + const top = row.problems[0] + const rest = row.problems.length - 1 + const p = row.protection + const ns = row.namespace + const radarFindings = row.problems.some((x) => x.severity !== 'posture') + + return ( +
+ {top && ( + 0 ? problemsLink?.(row.problems.length) ?? +{rest} more : null} + /> + )} + + {actions && actions.length > 0 && ( +
+ {actions.map((a) => ( + + ))} +
+ )} + + State + + + + + {row.controllerStatus.text} + + + {radarFindings ? 'reported by CNPG · Radar findings above are separate' : 'reported by CNPG'} + + + + +
+ + {row.instances.ready ?? '–'}/{row.instances.desired ?? '–'} ready + {row.cluster?.status?.currentPrimary && ( + · primary {row.cluster.status.currentPrimary} + )} + + {row.pods.length > 0 && ( +
+ {row.pods.map((pod) => ( + + ))} +
+ )} +
+
+ + + + + {row.replicaCluster && ( + + Follows {row.replicaCluster.source ? {row.replicaCluster.source} : 'an external primary'} + + )} + + {row.pgVersion ?? 'Unknown'} + {row.catalog && ( + + {' · '} + + {row.catalog.name} + + + )} + + + + + + {row.poolers.length === 0 ? ( + + {row.poolersKnown ? 'None' : 'No access to Poolers'} + + ) : ( + + {row.poolers.map((name) => ( + + ))} + + )} + + {row.gitops && ( + + {row.gitops.tool === 'argocd' ? 'Argo CD' : 'Flux'} {row.gitops.name} + + )} +
+ + Protection + + + + {p.schedule.names.length > 0 && ( +
+ {p.schedule.names.map((name) => ( + + ))} +
+ )} +
+ + {p.destination.objectStore ? ( + + ObjectStore {p.destination.objectStore} + + ) : ( + + )} + + + + + + + + + + {p.recoveryWindow.from ? ( +
+ + from {new Date(p.recoveryWindow.from).toUTCString().replace(' GMT', ' UTC')} {p.recoveryWindow.tone === 'degraded' ? '· not advancing' : 'to the newest archived WAL'} + + + {p.recoveryWindow.tone === 'degraded' && ( +
WAL archiving is failing, so nothing written since the last archived WAL can be recovered.
+ )} +
+ ) : ( + + )} +
+ + {p.restoreValidation.restoredInto ? ( + + {p.restoreValidation.text} + + ) : ( + + )} + + +
+ {extra} +
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx new file mode 100644 index 0000000000..dc667382d1 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx @@ -0,0 +1,264 @@ +import type { ReactNode } from 'react' +import { getCNPGDeclarativeMessage, getCNPGReclaimPolicy } from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' +import { ClusterLink, NotReported, ObjectProblems, SummaryShell } from './CNPGSharedSummary' +import { + appliedFact, + clustersIn, + databaseForDeclaration, + gitopsSourceOf, + missingManagedRole, + observedGenerationFact, + refOf, + relationUnavailable, + replicationForDatabase, + targetCluster, + workspaceList, +} from './relations' + +interface SummaryProps { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +} + +function ReclaimRow({ resource }: { resource: any }) { + const reclaim = getCNPGReclaimPolicy(resource) + return ( + + {reclaim.destructive ? ( + {reclaim.value} · removing this resource drops it from PostgreSQL + ) : ( + {reclaim.value} · removing this resource leaves PostgreSQL untouched + )} + + ) +} + +function Reconciled({ resource, extra }: { resource: any; extra?: ReactNode }) { + const message = getCNPGDeclarativeMessage(resource) + const applied = appliedFact(resource) + return ( + <> + Reconciled + + + + + + + + + {message ? ( + {message} + ) : ( + + )} + + {extra} + + + ) +} + +function DeclaredIn({ resource }: { resource: any }) { + const src = gitopsSourceOf(resource) + if (!src) return + return ( + + {src.tool === 'argocd' ? 'Argo CD application' : 'Flux'} {src.namespace ? `${src.namespace}/${src.name}` : src.name} + + ) +} + +function DatabaseRef({ resource, workspace, onNavigate }: SummaryProps) { + const dbname = resource?.spec?.dbname + if (!dbname) return + const ns = resource?.metadata?.namespace ?? '' + const db = relationUnavailable(workspace, 'databases', ns, 'Databases') + ? null + : databaseForDeclaration(resource, workspaceList(workspace, 'databases')) + return ( + + {dbname} + {db && ( + + {' · declared by Database '} + + + )} + + ) +} + +function LinkList({ items, kind, onNavigate }: { items: any[]; kind: string; onNavigate?: CNPGNavigate }) { + return ( + + {items.map((o) => ( + + ))} + + ) +} + +export function CNPGDatabaseSummary({ resource, workspace, onNavigate }: SummaryProps) { + const ns = resource?.metadata?.namespace ?? '' + const cluster = targetCluster(resource, clustersIn(workspace)) + const missingRole = missingManagedRole(resource, cluster) + const pubsUnavailable = relationUnavailable(workspace, 'publications', ns, 'Publications') + const subsUnavailable = relationUnavailable(workspace, 'subscriptions', ns, 'Subscriptions') + const related = replicationForDatabase(resource, workspaceList(workspace, 'publications'), workspaceList(workspace, 'subscriptions')) + + return ( + + + + Declared + + + {resource?.spec?.name ? {resource.spec.name} : } + + + {resource?.spec?.owner ? {resource.spec.owner} : } + + {resource?.spec?.ensure ?? 'present'} + + + + + “{missingRole}” is not among {cluster?.metadata?.name}'s managed roles + + ) + } + /> + + Source and target + + + + + + + + + {pubsUnavailable ? ( + + ) : related.publications.length === 0 ? ( + None on this database + ) : ( + + )} + + + {subsUnavailable ? ( + + ) : related.subscriptions.length === 0 ? ( + None on this database + ) : ( + + )} + + + + ) +} + +function publicationTargets(resource: any): ReactNode { + const target = resource?.spec?.target + if (target?.allTables === true) return 'All tables' + const objects = Array.isArray(target?.objects) ? target.objects : [] + if (objects.length === 0) return + const labels = objects.map((o: any) => { + if (o?.tablesInSchema) return `All tables in schema ${o.tablesInSchema}` + const t = o?.table + if (t?.name) { + const name = t.schema ? `${t.schema}.${t.name}` : t.name + return Array.isArray(t.columns) && t.columns.length > 0 ? `${name} (${t.columns.join(', ')})` : name + } + return 'Unrecognized entry' + }) + return ( +
    + {labels.map((l: string, i: number) => ( +
  • {l}
  • + ))} +
+ ) +} + +export function CNPGPublicationSummary({ resource, workspace, onNavigate }: SummaryProps) { + return ( + + + + Declared + + + {resource?.spec?.name ? {resource.spec.name} : } + + + + + + + + {publicationTargets(resource)} + + + + + + + + + ) +} + +export function CNPGSubscriptionSummary({ resource, workspace, onNavigate }: SummaryProps) { + const pub = resource?.spec?.publicationName + const ext = resource?.spec?.externalClusterName + return ( + + + + Declared + + + {resource?.spec?.name ? {resource.spec.name} : } + + + + + + + + + {pub ? ( + + Publication {pub} + {ext ? ( + + {' on external cluster '} + {ext} + + ) : null} + + ) : ( + + )} + + + + + + + + + + ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx new file mode 100644 index 0000000000..c7e05904ee --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx @@ -0,0 +1,80 @@ +import { CNPG_GROUP, getCNPGImageCatalogEntries } from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' +import { NotReported, Note, ObjectProblems, SummaryShell } from './CNPGSharedSummary' +import { clustersIn, clustersUsingCatalog, refOf, relationUnavailable } from './relations' + +export function CNPGImageCatalogSummary({ + resource, + workspace, + onNavigate, +}: { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +}) { + const clusterScoped = resource?.kind === 'ClusterImageCatalog' + const ns = resource?.metadata?.namespace ?? '' + const entries = getCNPGImageCatalogEntries(resource) + const majors = new Set(entries.map((e) => e.major)) + // A ClusterImageCatalog's users can be in any namespace, so partial coverage + // lists what was read; an ImageCatalog's users are all in its own namespace. + const partial = clusterScoped && workspace?.coverage?.clusters?.state === 'partial' + const unavailable = partial ? null : relationUnavailable(workspace, 'clusters', clusterScoped ? undefined : ns, 'Clusters') + const users = unavailable ? [] : clustersUsingCatalog(resource, clustersIn(workspace)) + + return ( + + + + Images + {entries.length === 0 ? ( +
+ +
+ ) : ( + + {entries.map((e) => ( + + {e.image} + + ))} + + )} + + Used by + {unavailable ? ( +
+ +
+ ) : users.length === 0 ? ( +
No visible cluster uses this catalog
+ ) : ( + + {users.map((u) => ( + + {clusterScoped ? `${u.cluster.metadata?.namespace}/${u.cluster.metadata?.name}` : u.cluster.metadata?.name} + + } + > + {u.major === null ? ( + + ) : majors.has(u.major) ? ( + `Requests PostgreSQL ${u.major}` + ) : ( + Requests PostgreSQL {u.major} · not in this catalog + )} + + ))} + + )} + {!unavailable && partial && Only clusters in namespaces you can read are listed.} + {!unavailable && !partial && clusterScoped && ( + Among clusters you can see; clusters in namespaces you cannot read are not listed + )} +
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx new file mode 100644 index 0000000000..39e1556ea9 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx @@ -0,0 +1,176 @@ +import { + CNPG_BARMAN_OBJECTSTORE_GROUP, + CNPG_GROUP, + getCNPGObjectStoreCredentialSecret, + getCNPGObjectStoreDestination, + getCNPGObjectStoreProvider, + getCNPGObjectStoreRecoveryWindows, + getCNPGObjectStoreRetention, +} from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' +import { NotReported, Note, ObjectProblems, SummaryShell, TimeAgo } from './CNPGSharedSummary' +import { clustersIn, inferredObjectStoreHealth, refOf, relationUnavailable, usersOfObjectStore } from './relations' + +function utc(at: string | undefined): string { + if (!at || !Number.isFinite(Date.parse(at))) return 'unknown' + return new Date(at).toUTCString().replace(' GMT', ' UTC') +} + +export function CNPGObjectStoreSummary({ + resource, + workspace, + onNavigate, +}: { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +}) { + const ns = resource?.metadata?.namespace ?? '' + const clustersUnavailable = relationUnavailable(workspace, 'clusters', ns, 'Clusters') + const users = clustersUnavailable ? [] : usersOfObjectStore(resource, clustersIn(workspace)) + const health = inferredObjectStoreHealth(resource, users) + const clusterForServer = new Map(users.map((u) => [u.serverName, u.cluster?.metadata?.name as string])) + const windows = getCNPGObjectStoreRecoveryWindows(resource) + const destination = getCNPGObjectStoreDestination(resource) + const provider = getCNPGObjectStoreProvider(resource) + const secret = getCNPGObjectStoreCredentialSecret(resource) + const retention = getCNPGObjectStoreRetention(resource) + const clusterWord = users.length === 1 ? "1 cluster's" : `${users.length} clusters'` + + return ( + + + + Upload health + {clustersUnavailable ? ( +
+ +
+ ) : ( + <> +
+ +
+ {users.length > 0 && ( + Inferred from {clusterWord} WAL archiving and backup results — ObjectStore has no health status + )} + {health.evidence.length > 0 && ( +
+ + {health.evidence.map((e) => ( + }> +
+ +
+ {e.window ? ( + <> + Last backup success + {e.window.lastFailedBackupTime && ( + + {' · '}last failure + + )} + + ) : ( + + )} +
+
+
+ ))} +
+
+ )} + + )} + + Recovery window + {windows.length === 0 ? ( +
+ +
+ ) : ( + + {windows.map((w) => { + const cluster = clusterForServer.get(w.server) + return ( + + {w.server} + + ) : ( + {w.server} + ) + } + > +
+
+ First recoverability point + {w.firstRecoverabilityPoint ? utc(w.firstRecoverabilityPoint) : } +
+
+ Last successful backup + {w.lastSuccessfulBackupTime ? utc(w.lastSuccessfulBackupTime) : } +
+ {w.lastFailedBackupTime && ( +
+ Last failed backup + {utc(w.lastFailedBackupTime)} +
+ )} + {w.failingSinceLastSuccess && ( + {w.lastSuccessfulBackupTime ? 'A backup failed after the last recorded success.' : 'No successful backup recorded.'} + )} +
+
+ ) + })} +
+ )} + + Destination + + {destination !== '-' ? {destination} : } + {provider ?? } + + {secret ? ( + + Secret + + ) : ( + + )} + + {retention ?? } + + + Used by + {clustersUnavailable ? ( +
+ +
+ ) : users.length === 0 ? ( +
+ No visible cluster uses this store + {workspace?.coverage?.clusters?.state === 'partial' && ( + Clusters in namespaces you cannot read are not checked. + )} +
+ ) : ( +
+ {users.map((u) => ( + + ))} +
+ )} +
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx new file mode 100644 index 0000000000..b715f2d584 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx @@ -0,0 +1,236 @@ +import { describe, expect, it } from 'vitest' +import { renderToString } from 'react-dom/server' +import { CNPGBackupSummary, CNPGScheduledBackupSummary } from './CNPGBackupSummary' +import { CNPGObjectStoreSummary } from './CNPGObjectStoreSummary' +import { CNPGDatabaseSummary } from './CNPGDeclarativeSummary' +import { CNPGPoolerSummary } from './CNPGPoolerSummary' +import { CNPGImageCatalogSummary } from './CNPGImageCatalogSummary' +import { CNPG_WORKSPACE_KEYS, type CNPGWorkspaceKey, type CNPGWorkspaceResponse } from './workspace' + +const PG = 'postgresql.cnpg.io/v1' +const PLUGIN = 'barman-cloud.cloudnative-pg.io' +const nav = () => {} + +function ws(objects: Partial>, over: Partial = {}): CNPGWorkspaceResponse { + const coverage: CNPGWorkspaceResponse['coverage'] = {} + for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } + return { + installed: true, + context: 'c', + namespaces: ['pg'], + coverage: { ...coverage, ...(over.coverage ?? {}) }, + objects, + issues: over.issues ?? [], + audit: [], + backupsOmitted: over.backupsOmitted ?? 0, + } +} + +function text(html: string): string { + return html + .split(/<[^>]*>/) + .join('') + .split(''').join("'") + .split('"').join('"') + .split('&').join('&') +} + +const mainCluster = { + apiVersion: PG, + kind: 'Cluster', + metadata: { name: 'main', namespace: 'pg' }, + spec: { plugins: [{ name: PLUGIN, parameters: { barmanObjectName: 'store' } }], managed: { roles: [{ name: 'reporting' }] } }, + status: { conditions: [{ type: 'ContinuousArchiving', status: 'False', message: 'upload failed' }] }, +} + +describe('CNPGBackupSummary', () => { + const b = { + apiVersion: PG, + kind: 'Backup', + metadata: { name: 'main-20260901', namespace: 'pg', labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, + spec: { cluster: { name: 'main' }, method: 'plugin', pluginConfiguration: { name: PLUGIN } }, + status: { + phase: 'failed', + startedAt: '2026-09-01T00:00:00Z', + stoppedAt: '2026-09-01T00:02:30Z', + instanceID: { podName: 'main-2' }, + error: 'can not upload', + }, + } + + it('shows the outcome and relationships from the same status fields', () => { + const html = renderToString() + const t = text(html) + expect(t).toContain('Failed') + expect(t).toContain('took 2m 30s') + expect(t).toContain('main-2') + expect(t).toContain('can not upload') + expect(t).toContain('ScheduledBackup nightly') + expect(t).toContain("ObjectStore store · from the Cluster's current configuration") + expect(t).not.toContain('On demand') + }) + + it('does not link a same-name schedule that replaced the one that took the backup', () => { + const owned = { ...b, metadata: { ...b.metadata, ownerReferences: [{ apiVersion: PG, kind: 'ScheduledBackup', name: 'nightly', uid: 'old' }] } } + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg', uid: 'new' } } + const t = text(renderToString()) + expect(t).toContain('an earlier schedule of that name') + }) + + it('shows its own issues on top', () => { + const issues = [ + { id: 'i1', severity: 'critical' as const, kind: 'Backup', group: 'postgresql.cnpg.io', namespace: 'pg', name: 'main-20260901', reason: 'CNPGBackupFailed', message: 'Backup failed' }, + { id: 'i2', severity: 'critical' as const, kind: 'Backup', group: 'velero.io', namespace: 'pg', name: 'main-20260901', reason: 'VeleroBackupFailed', message: 'Velero backup failed' }, + ] + const t = text(renderToString()) + expect(t).toContain('Backup failed') + expect(t).not.toContain('Velero backup failed') + }) +}) + +describe('CNPGScheduledBackupSummary', () => { + it('lists owned runs and notes omitted history', () => { + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' }, spec: { cluster: { name: 'main' }, schedule: '0 0 0 * * *' } } + const run = { apiVersion: PG, kind: 'Backup', metadata: { name: 'run-1', namespace: 'pg', labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, status: { phase: 'completed', startedAt: '2026-09-01T00:00:00Z' } } + const t = text(renderToString()) + expect(t).toContain('0 0 0 * * *') + expect(t).toContain('seconds first') + expect(t).not.toContain('Daily') + expect(t).toContain('run-1') + expect(t).toContain('Completed') + expect(t).toContain('Backups older than 7 days are not listed') + }) + + it('says when Backups are not readable instead of listing none', () => { + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' }, spec: { cluster: { name: 'main' } } } + const t = text(renderToString()) + expect(t).toContain('No access to Backups') + expect(t).not.toContain('No Backups from this schedule') + }) +}) + +describe('CNPGObjectStoreSummary', () => { + const store = { + apiVersion: 'barmancloud.cnpg.io/v1', + kind: 'ObjectStore', + metadata: { name: 'store', namespace: 'pg' }, + spec: { + configuration: { + destinationPath: 's3://bucket/pg', + s3Credentials: { accessKeyId: { name: 's3-creds', key: 'ACCESS_KEY_ID' }, secretAccessKey: { name: 's3-creds', key: 'SECRET' } }, + }, + retentionPolicy: '30d', + }, + status: { + serverRecoveryWindow: { + main: { firstRecoverabilityPoint: '2026-09-01T00:00:00Z', lastSuccessfulBackupTime: '2026-09-02T00:00:00Z', lastFailedBackupTime: '2026-09-03T00:00:00Z' }, + }, + }, + } + + it('labels upload health as inferred and explains the stalled window', () => { + const t = text(renderToString()) + expect(t).toContain('inferred') + expect(t).toContain("Inferred from 1 cluster's WAL archiving and backup results — ObjectStore has no health status") + expect(t).toContain('Uploads failing') + expect(t).toContain('A backup failed after the last recorded success.') + expect(t).not.toContain('restorable') + expect(t).toContain('s3-creds') + expect(t).toContain('s3://bucket/pg') + }) + + it('never renders credential keys or Secret contents', () => { + const html = renderToString() + expect(html).not.toContain('ACCESS_KEY_ID') + expect(html).not.toContain('SECRET') + }) + + it('records a failure with no success without claiming a recovery point', () => { + const failing = { ...store, status: { serverRecoveryWindow: { main: { lastFailedBackupTime: '2026-09-03T00:00:00Z' } } } } + const t = text(renderToString()) + expect(t).toContain('No successful backup recorded.') + expect(t).not.toContain('restorable') + }) + + it('says when no visible cluster uses the store', () => { + const t = text(renderToString()) + expect(t).toContain('No visible cluster uses this store') + expect(t).not.toContain('Inferred from') + }) +}) + +describe('CNPGDatabaseSummary', () => { + const db = (status: any) => ({ + apiVersion: PG, + kind: 'Database', + metadata: { name: 'app-db', namespace: 'pg', generation: 2 }, + spec: { cluster: { name: 'main' }, name: 'app', owner: 'app', databaseReclaimPolicy: 'delete' }, + status, + }) + + it('reports an unreconciled database as pending, not failed', () => { + const t = text(renderToString()) + expect(t).toContain('Pending') + expect(t).not.toContain('Not applied') + expect(t).toContain('drops it from PostgreSQL') + expect(t).toContain('GitOps source not recorded') + expect(t).not.toContain('Applied directly') + }) + + it('reports a failure and the missing managed role beside it', () => { + const t = text( + renderToString( + , + ), + ) + expect(t).toContain('Not applied') + expect(t).toContain('“app” is not among main\'s managed roles') + }) +}) + +describe('CNPGPoolerSummary', () => { + it('reports unknown scheduled count and unmeasured pressure', () => { + const pooler = { apiVersion: PG, kind: 'Pooler', metadata: { name: 'main-rw', namespace: 'pg' }, spec: { cluster: { name: 'main' }, type: 'rw', instances: 2 } } + const t = text(renderToString()) + expect(t).toContain('Scheduled count not reported') + expect(t).toContain('Not measured') + expect(t).toContain('main-rw') + }) +}) + +describe('CNPGImageCatalogSummary', () => { + it('lists users and flags a major the catalog lacks', () => { + const catalog = { apiVersion: PG, kind: 'ClusterImageCatalog', metadata: { name: 'pg' }, spec: { images: [{ major: 16, image: 'ghcr.io/cnpg/postgresql:16' }] } } + const clusters = [ + { apiVersion: PG, kind: 'Cluster', metadata: { name: 'a', namespace: 'x' }, spec: { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 16 } } }, + { apiVersion: PG, kind: 'Cluster', metadata: { name: 'b', namespace: 'y' }, spec: { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 17 } } }, + ] + const t = text(renderToString()) + expect(t).toContain('x/a') + expect(t).toContain('Requests PostgreSQL 17 · not in this catalog') + expect(t).toContain('clusters in namespaces you cannot read are not listed') + }) + + it('keeps known users when cluster coverage is partial', () => { + const catalog = { apiVersion: PG, kind: 'ClusterImageCatalog', metadata: { name: 'pg' }, spec: { images: [{ major: 16, image: 'img:16' }] } } + const clusters = [{ apiVersion: PG, kind: 'Cluster', metadata: { name: 'a', namespace: 'x' }, spec: { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 16 } } }] + const t = text( + renderToString( + , + ), + ) + expect(t).toContain('x/a') + expect(t).toContain('Only clusters in namespaces you can read are listed.') + expect(t).not.toContain('No access to Clusters') + }) + + it('reserves the unavailable text for denied coverage', () => { + const catalog = { apiVersion: PG, kind: 'ClusterImageCatalog', metadata: { name: 'pg' }, spec: {} } + const t = text(renderToString()) + expect(t).toContain('No access to Clusters') + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx new file mode 100644 index 0000000000..9418de0626 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx @@ -0,0 +1,70 @@ +import { getCNPGPoolerDeploymentName, getCNPGPoolerMode, getCNPGPoolerStatus, isCNPGPoolerPaused } from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, type CNPGNavigate } from './primitives' +import { ClusterLink, NotReported, Note, ObjectProblems, PhaseBadge, SummaryShell } from './CNPGSharedSummary' +import { refOf } from './relations' + +const TYPE_LABEL: Record = { + rw: 'rw · routes to the primary', + ro: 'ro · routes to replicas', + r: 'r · routes to any instance', +} + +export function CNPGPoolerSummary({ + resource, + workspace, + onNavigate, +}: { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +}) { + const ns = resource?.metadata?.namespace ?? '' + const type = resource?.spec?.type + const desired = resource?.spec?.instances + const scheduled = resource?.status?.instances + const deployment = getCNPGPoolerDeploymentName(resource) + + return ( + + + + State + + + + {isCNPGPoolerPaused(resource) && PgBouncer is paused: it holds client connections instead of serving them.} + + + + {typeof scheduled === 'number' ? `${scheduled} scheduled` : } + + {' · '} + {typeof desired === 'number' ? `${desired} desired` : 'desired not set'} + + + The Pooler counts scheduled pods, not ready ones; readiness is on its Deployment. + + + {deployment ? ( + + ) : ( + + )} + + + + + + + Routing + + + + + {type ? TYPE_LABEL[type] ?? type : } + {getCNPGPoolerMode(resource)} + + + ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx new file mode 100644 index 0000000000..301ee5ad23 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx @@ -0,0 +1,88 @@ +import type { ReactNode } from 'react' +import { Badge } from '../ui/Badge' +import type { StatusBadge as StatusBadgeValue } from '../resources/resource-utils' +import { CNPG_GROUP } from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceIssue, CNPGWorkspaceResponse } from './workspace' +import { FactValue, ProblemCallout, RefLink, type CNPGNavigate } from './primitives' +import { clustersIn, healthSeverity, problemsForObject, relationUnavailable, targetCluster, type CNPGObjectRef } from './relations' + +const MAX_PROBLEMS = 3 + +export function SummaryShell({ children }: { children: ReactNode }) { + return
{children}
+} + +/** The object's own Radar issues, most severe first. */ +export function ObjectProblems({ + issues, + subject, + onNavigate, +}: { + issues: CNPGWorkspaceIssue[] | undefined + subject: CNPGObjectRef + onNavigate?: CNPGNavigate +}) { + const problems = problemsForObject(issues, subject) + if (problems.length === 0) return null + const shown = problems.slice(0, MAX_PROBLEMS) + const rest = problems.length - shown.length + return ( +
+ {shown.map((p, i) => ( + 0 ? +{rest} more : null} + /> + ))} +
+ ) +} + +export function PhaseBadge({ status }: { status: StatusBadgeValue }) { + return ( + + {status.text} + + ) +} + +export function NotReported({ text = 'Not reported' }: { text?: string }) { + return {text} +} + +/** A timestamp as an age, with the absolute time on hover. */ +export function TimeAgo({ at, missing }: { at: unknown; missing?: string }) { + if (typeof at !== 'string' || !at || !Number.isFinite(Date.parse(at))) return + return +} + +export function Note({ children }: { children: ReactNode }) { + return
{children}
+} + +/** The Cluster an object declares itself against, linked. */ +export function ClusterLink({ + resource, + workspace, + onNavigate, +}: { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +}) { + const name = resource?.spec?.cluster?.name + if (!name) return + const ns = resource?.metadata?.namespace ?? '' + const visible = !!targetCluster(resource, clustersIn(workspace)) + return ( + + + {workspace && !visible && !relationUnavailable(workspace, 'clusters', ns, 'Clusters') && ( + · not found in this namespace + )} + + ) +} diff --git a/packages/k8s-ui/src/components/cnpg/index.ts b/packages/k8s-ui/src/components/cnpg/index.ts new file mode 100644 index 0000000000..08bd46e1fb --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/index.ts @@ -0,0 +1,14 @@ +export * from './workspace' +export * from './primitives' +export * from './CNPGClusterSummary' +export * from './CNPGBackupSummary' +export * from './CNPGObjectStoreSummary' +export * from './CNPGDeclarativeSummary' +export * from './CNPGPoolerSummary' +export * from './CNPGImageCatalogSummary' +export { + inferredObjectStoreHealth, + usersOfObjectStore, + type CNPGObjectStoreHealth, + type CNPGObjectStoreUser, +} from './relations' diff --git a/packages/k8s-ui/src/components/cnpg/primitives.tsx b/packages/k8s-ui/src/components/cnpg/primitives.tsx new file mode 100644 index 0000000000..7020a4da81 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/primitives.tsx @@ -0,0 +1,134 @@ +import type { ReactNode } from 'react' +import { clsx } from 'clsx' +import type { HealthLevel } from '../resources/resource-utils' +import { formatAge } from '../resources/resource-utils' +import { StatusDot } from '../ui/status-tone' +import { Tooltip } from '../ui/Tooltip' +import { AlertBanner } from '../ui/drawer-components' +import { TONE_TEXT_CLASS } from '../ui/severity-tone' +import type { CNPGFact, CNPGProblem } from './workspace' + +export interface CNPGRef { + kind: string + group?: string + namespace: string + name: string +} + +export type CNPGNavigate = (ref: CNPGRef) => void + +const TONE_TEXT: Record = { + healthy: 'text-theme-text-primary', + degraded: TONE_TEXT_CLASS.amber, + alert: TONE_TEXT_CLASS.orange, + unhealthy: TONE_TEXT_CLASS.red, + unknown: 'text-theme-text-tertiary', + neutral: 'text-theme-text-secondary', +} + +export const CNPG_PRIMARY_BUTTON = 'btn-brand inline-flex items-center gap-1.5 px-3 py-1.5 text-sm font-medium' +export const CNPG_SECONDARY_BUTTON = + 'inline-flex items-center gap-1.5 rounded-lg border border-theme-border bg-theme-surface px-3 py-1.5 text-sm text-theme-text-primary transition-colors hover:bg-theme-hover' + +export function toneTextClass(tone: HealthLevel): string { + return TONE_TEXT[tone] +} + +export function FactValue({ fact, className }: { fact: CNPGFact; className?: string }) { + const age = fact.at ? formatAge(fact.at) : null + const body = ( + + {fact.text} + {age && {fact.text ? ' · ' : ''}{age} ago} + + ) + if (!fact.source && !fact.at) return body + return ( + + {body} + + ) +} + +export function FactSource({ fact }: { fact: CNPGFact }) { + if (!fact.source) return null + return
{fact.source}
+} + +export function FactGrid({ children }: { children: ReactNode }) { + return
{children}
+} + +export function FactRow({ label, children }: { label: ReactNode; children: ReactNode }) { + return ( + <> +
{label}
+
{children}
+ + ) +} + +export function SummaryHeading({ children, hint }: { children: ReactNode; hint?: ReactNode }) { + return ( +
+

{children}

+ {hint && {hint}} +
+ ) +} + +export function RefLink({ refTo, onNavigate, children, mono }: { refTo: CNPGRef; onNavigate?: CNPGNavigate; children?: ReactNode; mono?: boolean }) { + const label = children ?? refTo.name + if (!onNavigate) return {label} + return ( + + ) +} + +const PROBLEM_VARIANT: Record = { + critical: 'error', + warning: 'warning', + posture: 'info', +} + +export function ProblemCallout({ + problem, + more, + onNavigate, + action, + subjectIsSelf, +}: { + problem: CNPGProblem + more?: ReactNode + onNavigate?: CNPGNavigate + action?: ReactNode + /** The callout sits on the subject's own page, so linking to it would loop. */ + subjectIsSelf?: boolean +}) { + const aboutChild = !subjectIsSelf && problem.subject.kind !== 'Cluster' + return ( + +
+ {aboutChild && ( + + {problem.subject.kind}{' '} + + + )} + {problem.source === 'audit' ? 'Radar check' : 'Radar issue'} + {action} + {more} +
+
+ ) +} + +export function ToneDot({ tone }: { tone: HealthLevel }) { + return +} diff --git a/packages/k8s-ui/src/components/cnpg/relations.test.ts b/packages/k8s-ui/src/components/cnpg/relations.test.ts new file mode 100644 index 0000000000..437dafbc70 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/relations.test.ts @@ -0,0 +1,304 @@ +import { describe, expect, it } from 'vitest' +import { + appliedFact, + backupDestination, + backupsForScheduledBackup, + clustersUsingCatalog, + databaseForDeclaration, + gitopsSourceOf, + inferredObjectStoreHealth, + isBackupFromSchedule, + issuesForObject, + missingManagedRole, + objectStoreForBackup, + relationUnavailable, + replicationForDatabase, + scheduledBackupOf, + usersOfObjectStore, +} from './relations' +import { CNPG_WORKSPACE_KEYS, type CNPGWorkspaceIssue, type CNPGWorkspaceResponse } from './workspace' + +const PG = 'postgresql.cnpg.io/v1' +const BARMAN = 'barmancloud.cnpg.io/v1' +const VELERO = 'velero.io/v1' +const PLUGIN = 'barman-cloud.cloudnative-pg.io' + +function cluster(name: string, ns = 'pg', spec: any = {}, status: any = {}): any { + return { apiVersion: PG, kind: 'Cluster', metadata: { name, namespace: ns }, spec, status } +} + +function pluginCluster(name: string, store: string, serverName?: string, conditions?: any[]): any { + return cluster( + name, + 'pg', + { plugins: [{ name: PLUGIN, isWALArchiver: true, parameters: { barmanObjectName: store, ...(serverName ? { serverName } : {}) } }] }, + conditions ? { conditions } : {}, + ) +} + +function backup(name: string, extra: any = {}): any { + return { + apiVersion: PG, + kind: 'Backup', + metadata: { name, namespace: 'pg', ...(extra.metadata ?? {}) }, + spec: { cluster: { name: 'main' }, ...(extra.spec ?? {}) }, + status: extra.status ?? {}, + } +} + +describe('scheduledBackupOf / backupsForScheduledBackup', () => { + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg', uid: 'uid-new' } } + const ownedBy = (name: string, uid: string, extra: any = {}) => + backup(name, { + ...extra, + metadata: { ...(extra.metadata ?? {}), ownerReferences: [{ apiVersion: PG, kind: 'ScheduledBackup', name: 'nightly', uid }] }, + }) + + it('reads the owner reference name', () => { + expect(scheduledBackupOf(ownedBy('b1', 'uid-new'))).toBe('nightly') + }) + + it('falls back to the operator label when the owner is the Cluster or unset', () => { + const b = backup('b1', { + metadata: { + labels: { 'cnpg.io/scheduled-backup': 'nightly' }, + ownerReferences: [{ apiVersion: PG, kind: 'Cluster', name: 'main' }], + }, + }) + expect(scheduledBackupOf(b)).toBe('nightly') + expect(isBackupFromSchedule(b, sched)).toBe(true) + }) + + it('returns null for an on-demand backup', () => { + expect(scheduledBackupOf(backup('b1'))).toBeNull() + }) + + it('ignores an owner reference from another API group', () => { + const b = backup('b1', { metadata: { ownerReferences: [{ apiVersion: 'example.com/v1', kind: 'ScheduledBackup', name: 'nightly' }] } }) + expect(scheduledBackupOf(b)).toBeNull() + }) + + it('does not attribute a recreated schedule the old schedule\'s Backups', () => { + const old = ownedBy('old', 'uid-old', { metadata: { labels: { 'cnpg.io/scheduled-backup': 'nightly' } } }) + expect(isBackupFromSchedule(old, sched)).toBe(false) + expect(isBackupFromSchedule(ownedBy('cur', 'uid-new'), sched)).toBe(true) + }) + + it('lists owned backups newest first and never a Velero Backup', () => { + const owned = (name: string, startedAt: string, apiVersion = PG) => ({ + ...backup(name, { metadata: { labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, status: { startedAt } }), + apiVersion, + }) + const list = [ + owned('old', '2026-09-01T00:00:00Z'), + owned('new', '2026-09-02T00:00:00Z'), + owned('velero', '2026-09-03T00:00:00Z', VELERO), + ownedBy('stale', 'uid-old', { status: { startedAt: '2026-09-04T00:00:00Z' } }), + backup('other'), + { ...owned('elsewhere', '2026-09-03T00:00:00Z'), metadata: { name: 'elsewhere', namespace: 'x', labels: { 'cnpg.io/scheduled-backup': 'nightly' } } }, + ] + expect(backupsForScheduledBackup(sched, list).map((b) => b.metadata.name)).toEqual(['new', 'old']) + }) +}) + +describe('objectStoreForBackup / backupDestination', () => { + const clusters = [pluginCluster('main', 'store-a')] + + it('prefers the store the backup recorded', () => { + const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN, parameters: { barmanObjectName: 'store-b' } } } }) + expect(objectStoreForBackup(b, clusters)).toEqual({ name: 'store-b', inferred: false }) + }) + + it("marks a store taken from the Cluster's current plugin as inferred", () => { + const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN } } }) + expect(objectStoreForBackup(b, clusters)).toEqual({ name: 'store-a', inferred: true }) + expect(backupDestination(b, clusters)).toEqual({ type: 'objectStore', name: 'store-a', inferred: true }) + }) + + it('checks plugin identity before reading barmanObjectName', () => { + const other = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: 'other.example.com', parameters: { barmanObjectName: 'store-b' } } } }) + expect(objectStoreForBackup(other, clusters)).toBeNull() + const unnamed = backup('b', { spec: { method: 'plugin', pluginConfiguration: { parameters: { barmanObjectName: 'store-b' } } } }) + expect(objectStoreForBackup(unnamed, clusters)).toBeNull() + }) + + it('reports in-tree paths and volume snapshots', () => { + expect(backupDestination(backup('b', { status: { method: 'barmanObjectStore', destinationPath: 's3://x' } }), clusters)).toEqual({ type: 'path', path: 's3://x' }) + expect(backupDestination(backup('b', { spec: { method: 'volumeSnapshot' } }), clusters)).toEqual({ type: 'volumeSnapshot' }) + expect(backupDestination(backup('b'), clusters)).toEqual({ type: 'unknown' }) + }) +}) + +describe('ObjectStore users and inferred health', () => { + const store = { + apiVersion: BARMAN, + kind: 'ObjectStore', + metadata: { name: 'store', namespace: 'pg' }, + status: { + serverRecoveryWindow: { + 'srv-a': { firstRecoverabilityPoint: '2026-09-01T00:00:00Z', lastSuccessfulBackupTime: '2026-09-02T00:00:00Z' }, + b: { lastSuccessfulBackupTime: '2026-09-02T00:00:00Z', lastFailedBackupTime: '2026-09-03T00:00:00Z' }, + }, + }, + } + const ok = [{ type: 'ContinuousArchiving', status: 'True' }] + + it('finds users in the same namespace, keyed by serverName', () => { + const clusters = [ + pluginCluster('a', 'store', 'srv-a', ok), + pluginCluster('b', 'store'), + pluginCluster('c', 'other'), + { ...pluginCluster('d', 'store'), metadata: { name: 'd', namespace: 'elsewhere' } }, + { ...pluginCluster('e', 'store'), apiVersion: 'cluster.x-k8s.io/v1beta1' }, + ] + const users = usersOfObjectStore(store, clusters) + expect(users.map((u) => [u.cluster.metadata.name, u.serverName])).toEqual([ + ['a', 'srv-a'], + ['b', 'b'], + ]) + }) + + it('reports failing when any user has a failure newer than its last success', () => { + const users = usersOfObjectStore(store, [pluginCluster('a', 'store', 'srv-a', ok), pluginCluster('b', 'store', undefined, ok)]) + const h = inferredObjectStoreHealth(store, users) + expect(h.summary.tone).toBe('unhealthy') + expect(h.summary.text).toBe('Uploads failing for 1 of 2 clusters') + expect(h.summary.at).toBe('2026-09-03T00:00:00Z') + }) + + it('reports failing from a False ContinuousArchiving condition', () => { + const users = usersOfObjectStore(store, [pluginCluster('a', 'store', 'srv-a', [{ type: 'ContinuousArchiving', status: 'False', message: 'boom' }])]) + expect(inferredObjectStoreHealth(store, users).summary).toMatchObject({ text: 'Uploads failing', tone: 'unhealthy' }) + }) + + it('is healthy only when every user reports archiving', () => { + const users = usersOfObjectStore(store, [pluginCluster('a', 'store', 'srv-a', ok)]) + expect(inferredObjectStoreHealth(store, users).summary.tone).toBe('healthy') + const unreported = usersOfObjectStore(store, [pluginCluster('a', 'store', 'srv-a')]) + expect(inferredObjectStoreHealth(store, unreported).summary).toMatchObject({ text: 'No failures reported', tone: 'unknown' }) + }) + + it('says so when nothing uses the store', () => { + expect(inferredObjectStoreHealth(store, []).summary).toMatchObject({ text: 'No visible cluster uses this store', tone: 'unknown' }) + }) +}) + +describe('clustersUsingCatalog', () => { + const catalog = { apiVersion: PG, kind: 'ImageCatalog', metadata: { name: 'pg', namespace: 'pg' } } + const clusterCatalog = { apiVersion: PG, kind: 'ClusterImageCatalog', metadata: { name: 'pg' } } + + it('treats an omitted ref kind as ImageCatalog and stays in the namespace', () => { + const clusters = [ + cluster('a', 'pg', { imageCatalogRef: { name: 'pg', major: 16 } }), + cluster('b', 'pg', { imageCatalogRef: { kind: 'ImageCatalog', name: 'pg', major: 17 } }), + cluster('c', 'other', { imageCatalogRef: { name: 'pg', major: 16 } }), + cluster('d', 'pg', { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 16 } }), + ] + expect(clustersUsingCatalog(catalog, clusters).map((u) => [u.cluster.metadata.name, u.major])).toEqual([ + ['a', 16], + ['b', 17], + ]) + }) + + it('matches ClusterImageCatalog refs from any namespace, never an omitted kind', () => { + const clusters = [ + cluster('a', 'x', { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 16 } }), + cluster('b', 'y', { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg' } }), + cluster('c', 'pg', { imageCatalogRef: { name: 'pg', major: 16 } }), + ] + expect(clustersUsingCatalog(clusterCatalog, clusters).map((u) => [u.cluster.metadata.name, u.major])).toEqual([ + ['a', 16], + ['b', null], + ]) + }) +}) + +describe('issuesForObject', () => { + const issue = (over: Partial): CNPGWorkspaceIssue => ({ + id: Math.random().toString(), + severity: 'warning', + kind: 'Backup', + group: 'postgresql.cnpg.io', + namespace: 'pg', + name: 'b1', + reason: 'CNPGBackupFailed', + ...over, + }) + + it('matches kind, group, namespace and name, critical first', () => { + const issues = [ + issue({ id: 'w' }), + issue({ id: 'c', severity: 'critical' }), + issue({ id: 'velero', group: 'velero.io' }), + issue({ id: 'ns', namespace: 'other' }), + issue({ id: 'name', name: 'b2' }), + issue({ id: 'kind', kind: 'ScheduledBackup' }), + ] + const got = issuesForObject(issues, { kind: 'Backup', group: 'postgresql.cnpg.io', namespace: 'pg', name: 'b1' }) + expect(got.map((i) => i.id)).toEqual(['c', 'w']) + }) +}) + +describe('declarations', () => { + it('distinguishes pending from failed', () => { + expect(appliedFact({ status: {} }).tone).toBe('unknown') + expect(appliedFact({ status: { applied: false } }).text).toBe('Not applied') + expect(appliedFact({ status: { applied: true } }).text).toBe('Applied') + }) + + it('names a missing role only when it is absent from managed roles', () => { + const db = { status: { applied: false, message: 'role "app" does not exist' } } + expect(missingManagedRole(db, cluster('main', 'pg', { managed: { roles: [{ name: 'other' }] } }))).toBe('app') + expect(missingManagedRole(db, cluster('main', 'pg', { managed: { roles: [{ name: 'app' }] } }))).toBeNull() + expect(missingManagedRole(db, null)).toBeNull() + expect(missingManagedRole({ status: { message: 'connection refused' } }, cluster('main'))).toBeNull() + }) + + it('reads the GitOps owner labels', () => { + expect(gitopsSourceOf({ metadata: { labels: { 'argocd.argoproj.io/instance': 'app' } } })).toEqual({ tool: 'argocd', name: 'app' }) + expect(gitopsSourceOf({ metadata: { labels: { 'kustomize.toolkit.fluxcd.io/name': 'k', 'kustomize.toolkit.fluxcd.io/namespace': 'flux' } } })).toEqual({ + tool: 'flux', + name: 'k', + namespace: 'flux', + }) + expect(gitopsSourceOf({ metadata: {} })).toBeNull() + }) + + const decl = (kind: string, name: string, clusterName: string, dbname: string, ns = 'pg') => ({ + apiVersion: PG, + kind, + metadata: { name, namespace: ns }, + spec: { cluster: { name: clusterName }, dbname }, + }) + const database = { apiVersion: PG, kind: 'Database', metadata: { name: 'app-db', namespace: 'pg' }, spec: { cluster: { name: 'main' }, name: 'app' } } + + it('finds publications and subscriptions on the same cluster and database', () => { + const pubs = [decl('Publication', 'p1', 'main', 'app'), decl('Publication', 'p2', 'main', 'other'), decl('Publication', 'p3', 'other', 'app')] + const subs = [decl('Subscription', 's1', 'main', 'app'), decl('Subscription', 's2', 'main', 'app', 'x')] + const r = replicationForDatabase(database, pubs, subs) + expect(r.publications.map((p) => p.metadata.name)).toEqual(['p1']) + expect(r.subscriptions.map((s) => s.metadata.name)).toEqual(['s1']) + }) + + it('finds the Database declaring a publication database', () => { + expect(databaseForDeclaration(decl('Publication', 'p1', 'main', 'app'), [database])?.metadata.name).toBe('app-db') + expect(databaseForDeclaration(decl('Publication', 'p1', 'other', 'app'), [database])).toBeNull() + }) +}) + +describe('relationUnavailable', () => { + const ws = (state: any, deniedNamespaces?: string[]): CNPGWorkspaceResponse => { + const coverage: CNPGWorkspaceResponse['coverage'] = {} + for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } + coverage.backups = { state, deniedNamespaces } + return { installed: true, context: 'c', namespaces: null, coverage, objects: {}, issues: [], audit: [], backupsOmitted: 0 } + } + + it('answers from coverage', () => { + expect(relationUnavailable(ws('full'), 'backups', 'pg', 'Backups')).toBeNull() + expect(relationUnavailable(ws('partial', ['pg']), 'backups', 'pg', 'Backups')).toBe('No access to Backups') + expect(relationUnavailable(ws('partial', ['other']), 'backups', 'pg', 'Backups')).toBeNull() + expect(relationUnavailable(ws('denied'), 'backups', 'pg', 'Backups')).toBe('No access to Backups') + expect(relationUnavailable(null, 'backups', 'pg', 'Backups')).toBe('Backups could not be read') + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/relations.ts b/packages/k8s-ui/src/components/cnpg/relations.ts new file mode 100644 index 0000000000..2db316d662 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/relations.ts @@ -0,0 +1,417 @@ +// Pure relationship lookups between CloudNativePG objects in the workspace +// payload. Each helper answers only from what the objects record; a relation +// that cannot be established returns null or an empty list, never a guess. + +import type { BadgeSeverity } from '../ui/Badge' +import type { HealthLevel } from '../resources/resource-utils' +import { + CNPG_BARMAN_PLUGIN_NAME, + CNPG_GROUP, + getCNPGClusterBarmanPlugin, + getCNPGObjectStoreRecoveryWindows, + isApiGroup, + type CNPGObjectStoreRecoveryWindow, +} from '../resources/resource-utils-cnpg' +import { + cnpgIssueCategory, + coverageReadable, + type CNPGFact, + type CNPGProblem, + type CNPGWorkspaceIssue, + type CNPGWorkspaceKey, + type CNPGWorkspaceResponse, +} from './workspace' + +export interface CNPGObjectRef { + kind: string + group: string + namespace: string + name: string +} + +const SCHEDULED_BACKUP_LABEL = 'cnpg.io/scheduled-backup' + +function nsOf(obj: any): string { + return obj?.metadata?.namespace ?? '' +} + +function nameOf(obj: any): string { + return obj?.metadata?.name ?? '' +} + +function specCluster(obj: any): string | undefined { + const n = obj?.spec?.cluster?.name + return typeof n === 'string' && n ? n : undefined +} + +function parseTime(t: unknown): number { + const ms = typeof t === 'string' ? Date.parse(t) : NaN + return Number.isFinite(ms) ? ms : 0 +} + +/** Kind and API group both match. Kind alone is not identity: Velero also ships `Backup`. */ +export function isCNPGKind(obj: any, kind: string, group: string = CNPG_GROUP): boolean { + return obj?.kind === kind && isApiGroup(obj?.apiVersion, group) +} + +export function refOf(obj: any, kind: string, group: string = CNPG_GROUP): CNPGObjectRef { + return { kind, group, namespace: nsOf(obj), name: nameOf(obj) } +} + +export function healthSeverity(level: HealthLevel): BadgeSeverity { + switch (level) { + case 'healthy': + return 'success' + case 'unhealthy': + return 'error' + case 'alert': + return 'alert' + case 'degraded': + return 'warning' + default: + return 'neutral' + } +} + +// --------------------------------------------------------------------------- +// Workspace access +// --------------------------------------------------------------------------- + +export function workspaceList(ws: CNPGWorkspaceResponse | null | undefined, key: CNPGWorkspaceKey): any[] { + return ws?.objects?.[key] ?? [] +} + +/** + * Why the workspace cannot answer for `key` in `namespace`, or null when it can. + * A null workspace means the aggregate could not be read at all. + */ +export function relationUnavailable( + ws: CNPGWorkspaceResponse | null | undefined, + key: CNPGWorkspaceKey, + namespace: string | undefined, + what: string, +): string | null { + if (!ws) return `${what} could not be read` + const cov = ws.coverage?.[key] ?? { state: 'notInstalled' as const } + if (coverageReadable(cov, namespace)) return null + switch (cov.state) { + case 'denied': + case 'partial': + return `No access to ${what}` + case 'syncing': + return 'Loading…' + case 'error': + return `Could not read ${what}` + default: + return `${what} are not installed` + } +} + +export function clustersIn(ws: CNPGWorkspaceResponse | null | undefined): any[] { + return workspaceList(ws, 'clusters').filter((c) => isCNPGKind(c, 'Cluster')) +} + +/** The Cluster an object declares itself against (`spec.cluster.name`), if visible. */ +export function targetCluster(obj: any, clusters: any[]): any | null { + const name = specCluster(obj) + if (!name) return null + return clusters.find((c) => isCNPGKind(c, 'Cluster') && nsOf(c) === nsOf(obj) && nameOf(c) === name) ?? null +} + +// --------------------------------------------------------------------------- +// Issues +// --------------------------------------------------------------------------- + +export function issuesForObject(issues: CNPGWorkspaceIssue[] | undefined, ref: CNPGObjectRef): CNPGWorkspaceIssue[] { + const rank = { critical: 0, warning: 1 } as const + return (issues ?? []) + .filter( + (i) => + i.kind === ref.kind && + (i.group ?? '') === ref.group && + (i.namespace ?? '') === ref.namespace && + i.name === ref.name, + ) + .sort((a, b) => rank[a.severity] - rank[b.severity]) +} + +export function problemsForObject(issues: CNPGWorkspaceIssue[] | undefined, ref: CNPGObjectRef): CNPGProblem[] { + return issuesForObject(issues, ref).map((issue) => ({ + id: `${issue.id}:${issue.kind}/${issue.name}`, + severity: issue.severity, + category: cnpgIssueCategory(issue), + title: issue.message || issue.reason, + detail: issue.cause || undefined, + subject: { kind: issue.kind, group: issue.group ?? '', namespace: issue.namespace ?? '', name: issue.name }, + source: 'issue', + })) +} + +// --------------------------------------------------------------------------- +// Backups and schedules +// --------------------------------------------------------------------------- + +function scheduleOwnerRefs(backup: any): any[] { + const refs = backup?.metadata?.ownerReferences + if (!Array.isArray(refs)) return [] + return refs.filter((r: any) => r?.kind === 'ScheduledBackup' && isApiGroup(r?.apiVersion, CNPG_GROUP)) +} + +/** + * The name of the ScheduledBackup that created a Backup. The owner reference + * is only set when the schedule's `backupOwnerReference` is `self`; the + * operator labels every Backup it creates from a schedule regardless, so the + * label is the fallback when no such owner reference exists. + */ +export function scheduledBackupOf(backup: any): string | null { + const owner = scheduleOwnerRefs(backup)[0] + if (owner?.name) return owner.name + const label = backup?.metadata?.labels?.[SCHEDULED_BACKUP_LABEL] + return typeof label === 'string' && label ? label : null +} + +/** + * Whether this schedule created the Backup. An owner reference must match by + * uid: a schedule deleted and recreated under the same name did not create the + * old one's Backups. The label, which carries only a name, is read only when + * no ScheduledBackup owner reference exists. + */ +export function isBackupFromSchedule(backup: any, schedule: any): boolean { + if (!isCNPGKind(backup, 'Backup') || nsOf(backup) !== nsOf(schedule)) return false + const owners = scheduleOwnerRefs(backup) + if (owners.length > 0) { + const uid = schedule?.metadata?.uid + return owners.some((r: any) => r?.name === nameOf(schedule) && !!uid && r?.uid === uid) + } + return backup?.metadata?.labels?.[SCHEDULED_BACKUP_LABEL] === nameOf(schedule) +} + +/** Backups a ScheduledBackup created, newest first. */ +export function backupsForScheduledBackup(schedule: any, backups: any[]): any[] { + return backups.filter((b) => isBackupFromSchedule(b, schedule)).sort((a, b) => backupTime(b) - backupTime(a)) +} + +export function backupTime(backup: any): number { + return parseTime(backup?.status?.startedAt) || parseTime(backup?.metadata?.creationTimestamp) +} + +/** + * The ObjectStore a barman-cloud plugin Backup wrote to. The Backup's own + * plugin parameters are a record of that run; the target Cluster's plugin + * configuration is only what it is configured with now, so a store taken from + * there is marked inferred. + */ +export function objectStoreForBackup(backup: any, clusters: any[]): { name: string; inferred: boolean } | null { + const method = backup?.status?.method || backup?.spec?.method + if (method !== 'plugin') return null + const cfg = backup?.spec?.pluginConfiguration + if (cfg?.name !== CNPG_BARMAN_PLUGIN_NAME) return null + const own = cfg?.parameters?.barmanObjectName + if (typeof own === 'string' && own) return { name: own, inferred: false } + const cluster = targetCluster(backup, clusters) + const current = cluster ? getCNPGClusterBarmanPlugin(cluster)?.barmanObjectName : undefined + return current ? { name: current, inferred: true } : null +} + +export type CNPGBackupDestination = + | { type: 'objectStore'; name: string; inferred: boolean } + | { type: 'path'; path: string } + | { type: 'volumeSnapshot' } + | { type: 'unknown' } + +export function backupDestination(backup: any, clusters: any[]): CNPGBackupDestination { + const store = objectStoreForBackup(backup, clusters) + if (store) return { type: 'objectStore', ...store } + const method = backup?.status?.method || backup?.spec?.method + if (method === 'volumeSnapshot') return { type: 'volumeSnapshot' } + const path = backup?.status?.destinationPath + if (typeof path === 'string' && path) return { type: 'path', path } + return { type: 'unknown' } +} + +// --------------------------------------------------------------------------- +// ObjectStore +// --------------------------------------------------------------------------- + +export interface CNPGObjectStoreUser { + cluster: any + /** Key of this cluster's archive inside the store's recovery windows. */ + serverName: string +} + +/** Clusters in the store's namespace whose barman-cloud plugin archives to it. */ +export function usersOfObjectStore(store: any, clusters: any[]): CNPGObjectStoreUser[] { + const ns = nsOf(store) + const name = nameOf(store) + const out: CNPGObjectStoreUser[] = [] + for (const c of clusters) { + if (!isCNPGKind(c, 'Cluster') || nsOf(c) !== ns) continue + const plugin = getCNPGClusterBarmanPlugin(c) + if (plugin?.barmanObjectName !== name) continue + out.push({ cluster: c, serverName: plugin.serverName || nameOf(c) }) + } + return out.sort((a, b) => nameOf(a.cluster).localeCompare(nameOf(b.cluster))) +} + +export interface CNPGObjectStoreEvidence { + cluster: CNPGObjectRef + serverName: string + archiving: CNPGFact + window: CNPGObjectStoreRecoveryWindow | null +} + +export interface CNPGObjectStoreHealth { + summary: CNPGFact + evidence: CNPGObjectStoreEvidence[] +} + +function archivingFact(cluster: any): CNPGFact { + const conds = cluster?.status?.conditions + const c = Array.isArray(conds) ? conds.find((x: any) => x?.type === 'ContinuousArchiving') : null + if (!c) return { text: 'WAL archiving not reported', tone: 'unknown' } + if (c.status === 'True') return { text: 'WAL archiving', tone: 'healthy', at: c.lastTransitionTime } + if (c.status === 'False') { + return { text: c.message ? `WAL archiving failing · ${c.message}` : 'WAL archiving failing', tone: 'unhealthy', at: c.lastTransitionTime } + } + return { text: 'WAL archiving unknown', tone: 'unknown' } +} + +/** + * Upload health for an ObjectStore, inferred from the clusters that use it. + * ObjectStore publishes no health of its own, so the only evidence is each + * user cluster's ContinuousArchiving condition and whether the store's + * recovery window for that cluster records a failure newer than its last + * success. + */ +export function inferredObjectStoreHealth(store: any, users: CNPGObjectStoreUser[]): CNPGObjectStoreHealth { + const windows = getCNPGObjectStoreRecoveryWindows(store) + const evidence: CNPGObjectStoreEvidence[] = users.map((u) => ({ + cluster: refOf(u.cluster, 'Cluster'), + serverName: u.serverName, + archiving: archivingFact(u.cluster), + window: windows.find((w) => w.server === u.serverName) ?? null, + })) + if (evidence.length === 0) { + return { summary: { text: 'No visible cluster uses this store', tone: 'unknown' }, evidence } + } + const failing = evidence.filter((e) => e.archiving.tone === 'unhealthy' || e.window?.failingSinceLastSuccess) + if (failing.length > 0) { + const latestFailure = failing + .map((e) => e.window?.lastFailedBackupTime ?? (e.archiving.tone === 'unhealthy' ? e.archiving.at : undefined)) + .filter((t): t is string => !!t) + .sort((a, b) => parseTime(b) - parseTime(a))[0] + return { + summary: { + text: failing.length === evidence.length ? 'Uploads failing' : `Uploads failing for ${failing.length} of ${evidence.length} clusters`, + tone: 'unhealthy', + at: latestFailure, + }, + evidence, + } + } + if (evidence.every((e) => e.archiving.tone === 'healthy')) { + return { summary: { text: 'Uploads succeeding', tone: 'healthy' }, evidence } + } + return { summary: { text: 'No failures reported', tone: 'unknown' }, evidence } +} + +// --------------------------------------------------------------------------- +// Declarative objects +// --------------------------------------------------------------------------- + +export function appliedFact(obj: any): CNPGFact { + const applied = obj?.status?.applied + if (applied === true) return { text: 'Applied', tone: 'healthy' } + if (applied === false) return { text: 'Not applied', tone: 'unhealthy' } + return { text: 'Pending · the operator has not reported a result yet', tone: 'unknown' } +} + +export function observedGenerationFact(obj: any): CNPGFact { + const observed = obj?.status?.observedGeneration + const generation = obj?.metadata?.generation + if (typeof observed !== 'number') return { text: 'Not reported', tone: 'unknown' } + if (typeof generation !== 'number') return { text: `Generation ${observed}`, tone: 'neutral' } + if (observed >= generation) return { text: `Current · generation ${generation}`, tone: 'neutral' } + return { text: `Behind · observed ${observed}, declared ${generation}`, tone: 'degraded' } +} + +/** + * A role the operator says is missing, when that role is also absent from the + * target Cluster's managed roles. Only an adjacent fact: roles may be created + * outside `spec.managed.roles`. + */ +export function missingManagedRole(obj: any, cluster: any | null): string | null { + if (!cluster) return null + const msg = obj?.status?.message + if (typeof msg !== 'string') return null + const m = msg.match(/role "([^"]+)" does not exist/) + if (!m) return null + const roles = cluster?.spec?.managed?.roles + const names = Array.isArray(roles) ? roles.map((r: any) => r?.name) : [] + return names.includes(m[1]) ? null : m[1] +} + +export { cnpgGitOpsSource as gitopsSourceOf } from './workspace' + +/** Publications and Subscriptions on the same Cluster and PostgreSQL database. */ +export function replicationForDatabase( + database: any, + publications: any[], + subscriptions: any[], +): { publications: any[]; subscriptions: any[] } { + const ns = nsOf(database) + const cluster = specCluster(database) + const dbname = database?.spec?.name + const match = (o: any, kind: string) => + isCNPGKind(o, kind) && !!cluster && nsOf(o) === ns && specCluster(o) === cluster && !!dbname && o?.spec?.dbname === dbname + return { + publications: publications.filter((p) => match(p, 'Publication')), + subscriptions: subscriptions.filter((s) => match(s, 'Subscription')), + } +} + +/** The Database object declaring the PostgreSQL database a Publication/Subscription runs in. */ +export function databaseForDeclaration(obj: any, databases: any[]): any | null { + const cluster = specCluster(obj) + const dbname = obj?.spec?.dbname + if (!cluster || !dbname) return null + return ( + databases.find( + (d) => isCNPGKind(d, 'Database') && nsOf(d) === nsOf(obj) && specCluster(d) === cluster && d?.spec?.name === dbname, + ) ?? null + ) +} + +// --------------------------------------------------------------------------- +// Image catalogs +// --------------------------------------------------------------------------- + +export interface CNPGImageCatalogUser { + cluster: any + major: number | null +} + +/** + * Clusters pinned to a catalog through `spec.imageCatalogRef`. An ImageCatalog + * is namespace-local; a ClusterImageCatalog is referenceable from any + * namespace. A ref without `kind` means ImageCatalog. + */ +export function clustersUsingCatalog(catalog: any, clusters: any[]): CNPGImageCatalogUser[] { + const kind = catalog?.kind + if (kind !== 'ImageCatalog' && kind !== 'ClusterImageCatalog') return [] + if (!isApiGroup(catalog?.apiVersion, CNPG_GROUP)) return [] + const name = nameOf(catalog) + return clusters + .filter((c) => { + if (!isCNPGKind(c, 'Cluster')) return false + const ref = c?.spec?.imageCatalogRef + if (!ref?.name || ref.name !== name) return false + if ((ref.kind || 'ImageCatalog') !== kind) return false + return kind === 'ClusterImageCatalog' || nsOf(c) === nsOf(catalog) + }) + .map((c) => { + const major = c?.spec?.imageCatalogRef?.major + return { cluster: c, major: typeof major === 'number' ? major : null } + }) + .sort((a, b) => nsOf(a.cluster).localeCompare(nsOf(b.cluster)) || nameOf(a.cluster).localeCompare(nameOf(b.cluster))) +} diff --git a/packages/k8s-ui/src/components/cnpg/workspace.test.ts b/packages/k8s-ui/src/components/cnpg/workspace.test.ts new file mode 100644 index 0000000000..beeb5dd65b --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/workspace.test.ts @@ -0,0 +1,285 @@ +import { describe, it, expect } from 'vitest' +import { buildCNPGFleet, type CNPGWorkspaceResponse, type CNPGWorkspaceKey, CNPG_WORKSPACE_KEYS } from './workspace' + +const G = 'postgresql.cnpg.io/v1' + +function cluster(name: string, ns: string, extra: any = {}): any { + return { + apiVersion: G, + kind: 'Cluster', + metadata: { name, namespace: ns, ...(extra.metadata ?? {}) }, + spec: { instances: 3, ...(extra.spec ?? {}) }, + status: { + phase: 'Cluster in healthy state', + readyInstances: 3, + currentPrimary: `${name}-1`, + ...(extra.status ?? {}), + }, + } +} + +function pod(name: string, ns: string, clusterName: string, role: string, ready = true): any { + return { + apiVersion: 'v1', + kind: 'Pod', + metadata: { name, namespace: ns, labels: { 'cnpg.io/cluster': clusterName, 'cnpg.io/instanceRole': role } }, + status: { conditions: [{ type: 'Ready', status: ready ? 'True' : 'False' }] }, + } +} + +function resp(objects: Partial>, over: Partial = {}): CNPGWorkspaceResponse { + const coverage: CNPGWorkspaceResponse['coverage'] = {} + for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } + return { + installed: true, + context: 'test', + namespaces: null, + coverage: { ...coverage, ...(over.coverage ?? {}) }, + objects, + issues: over.issues ?? [], + audit: over.audit ?? [], + backupsOmitted: 0, + } +} + +describe('buildCNPGFleet', () => { + it('never reports replication as healthy from pod readiness alone', () => { + const fleet = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db')], + pods: [pod('pg-a-1', 'db', 'pg-a', 'primary'), pod('pg-a-2', 'db', 'pg-a', 'replica'), pod('pg-a-3', 'db', 'pg-a', 'replica')], + }), + ) + const row = fleet.rows[0] + expect(row.replication.tone).toBe('unknown') + expect(row.replication.text).toContain('2/2 replicas ready') + expect(row.pods[0].role).toBe('primary') + }) + + it('attributes child-object issues to their cluster and marks attention', () => { + const fleet = buildCNPGFleet( + resp( + { + clusters: [cluster('pg-a', 'db'), cluster('pg-b', 'db')], + databases: [{ apiVersion: G, kind: 'Database', metadata: { name: 'reporting', namespace: 'db' }, spec: { cluster: { name: 'pg-b' } }, status: { applied: false } }], + }, + { + issues: [ + { id: 'i1', severity: 'warning', kind: 'Database', group: 'postgresql.cnpg.io', namespace: 'db', name: 'reporting', reason: 'CNPGDeclarativeNotApplied', message: 'role "x" does not exist' }, + ], + }, + ), + ) + const b = fleet.rows.find((r) => r.name === 'pg-b')! + const a = fleet.rows.find((r) => r.name === 'pg-a')! + expect(b.attention).toBe(true) + expect(b.categories.has('declarations')).toBe(true) + expect(b.declarations.failed).toBe(1) + expect(a.attention).toBe(false) + expect(fleet.attentionCount).toBe(1) + expect(fleet.categoryCounts.declarations).toBe(1) + expect(fleet.rows[0].name).toBe('pg-b') + }) + + it('treats the no-schedule audit finding as posture, not attention, and words it narrowly', () => { + const fleet = buildCNPGFleet( + resp({ clusters: [cluster('pg-a', 'db')] }, { + audit: [{ checkId: 'cnpgNoDeclarativeBackup', severity: 'warning', kind: 'Cluster', namespace: 'db', name: 'pg-a', message: 'no ScheduledBackup' }], + }), + ) + const row = fleet.rows[0] + expect(row.attention).toBe(false) + expect(row.problems[0].title).toBe('No declarative backup schedule') + expect(row.protection.schedule.text).toBe('No declarative schedule') + }) + + it('says "no access" rather than "none" when a kind is not readable', () => { + const fleet = buildCNPGFleet( + resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'denied' }, scheduledBackups: { state: 'partial', deniedNamespaces: ['db'] } } }), + ) + const p = fleet.rows[0].protection + expect(p.lastSuccessfulBackup.text).toBe('No access to Backups') + expect(p.lastSuccessfulBackup.tone).toBe('unknown') + expect(p.schedule.text).toBe('No access to ScheduledBackups') + expect(fleet.incompleteKinds).toEqual(expect.arrayContaining(['backups', 'scheduledBackups'])) + }) + + it('prefers the newest successful backup across Backup CRs and ObjectStore status, citing the source', () => { + const fleet = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } })], + backups: [{ apiVersion: G, kind: 'Backup', metadata: { name: 'pg-a-old', namespace: 'db' }, spec: { cluster: { name: 'pg-a' } }, status: { phase: 'completed', stoppedAt: '2026-09-20T02:00:00Z' } }], + objectStores: [{ + apiVersion: 'barmancloud.cnpg.io/v1', kind: 'ObjectStore', metadata: { name: 'store', namespace: 'db' }, + status: { serverRecoveryWindow: { 'pg-a': { firstRecoverabilityPoint: '2026-09-01T00:00:00Z', lastSuccessfulBackupTime: '2026-09-28T02:00:00Z' } } }, + }], + }), + ) + const p = fleet.rows[0].protection + expect(p.lastSuccessfulBackup.at).toBe('2026-09-28T02:00:00Z') + expect(p.lastSuccessfulBackup.source).toBe('ObjectStore store status') + expect(p.recoveryWindow.from).toBe('2026-09-01T00:00:00Z') + expect(p.destination.method).toBe('plugin') + }) + + it('ignores deprecated Cluster status backup fields for plugin clusters', () => { + const fleet = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db', { + spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] }, + status: { lastSuccessfulBackup: '2020-01-01T00:00:00Z' }, + })], + }), + ) + expect(fleet.rows[0].protection.lastSuccessfulBackup.text).toBe('None observed') + }) + + it('never marks restore validation healthy', () => { + const src = cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } }) + const restored = cluster('pg-a-restore', 'db', { + spec: { + bootstrap: { recovery: { source: 'origin' } }, + externalClusters: [{ name: 'origin', plugin: { name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store', serverName: 'pg-a' } } }], + }, + }) + const fleet = buildCNPGFleet(resp({ clusters: [src, restored] })) + const a = fleet.rows.find((r) => r.name === 'pg-a')! + expect(a.protection.restoreValidation.text).toBe('Restored into pg-a-restore') + expect(a.protection.restoreValidation.tone).toBe('neutral') + const r = fleet.rows.find((x) => x.name === 'pg-a-restore')! + expect(r.protection.restoreValidation.text).toBe('None recorded') + expect(r.protection.restoreValidation.tone).toBe('unknown') + }) + + it('ends the recovery window at WAL archiving, not at the last base backup', () => { + const plugin = { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } + const store = { + apiVersion: 'barmancloud.cnpg.io/v1', kind: 'ObjectStore', metadata: { name: 'store', namespace: 'db' }, + status: { serverRecoveryWindow: { 'pg-a': { firstRecoverabilityPoint: '2026-09-01T00:00:00Z', lastSuccessfulBackupTime: '2026-09-02T00:00:00Z', lastFailedBackupTime: '2026-09-03T00:00:00Z' } } }, + } + const archiving = cluster('pg-a', 'db', { spec: plugin, status: { conditions: [{ type: 'ContinuousArchiving', status: 'True' }] } }) + const w = buildCNPGFleet(resp({ clusters: [archiving], objectStores: [store] })).rows[0].protection.recoveryWindow + expect(w).toMatchObject({ from: '2026-09-01T00:00:00Z', tone: 'neutral' }) + expect(w).not.toHaveProperty('to') + const failing = cluster('pg-a', 'db', { spec: plugin, status: { conditions: [{ type: 'ContinuousArchiving', status: 'False' }] } }) + expect(buildCNPGFleet(resp({ clusters: [failing], objectStores: [store] })).rows[0].protection.recoveryWindow.tone).toBe('degraded') + }) + + it('reports WAL archiving from the condition and unknown when absent', () => { + const failing = cluster('pg-a', 'db', { status: { conditions: [{ type: 'ContinuousArchiving', status: 'False', message: 'exit status 1' }] } }) + const silent = cluster('pg-b', 'db') + const fleet = buildCNPGFleet(resp({ clusters: [failing, silent] })) + expect(fleet.rows.find((r) => r.name === 'pg-a')!.protection.walArchiving.tone).toBe('unhealthy') + expect(fleet.rows.find((r) => r.name === 'pg-a')!.protection.summary.text).toBe('WAL archiving failing') + expect(fleet.rows.find((r) => r.name === 'pg-b')!.protection.walArchiving.tone).toBe('unknown') + }) + + it('ignores same-named kinds from other API groups', () => { + const capi = { apiVersion: 'cluster.x-k8s.io/v1beta1', kind: 'Cluster', metadata: { name: 'workload', namespace: 'db' } } + const fleet = buildCNPGFleet(resp({ clusters: [capi, cluster('pg-a', 'db')] })) + expect(fleet.rows.map((r) => r.name)).toEqual(['pg-a']) + }) + + it('counts declared managed roles and their reconcile errors', () => { + const c = cluster('pg-a', 'db', { + spec: { managed: { roles: [{ name: 'app' }, { name: 'audit' }] } }, + status: { managedRolesStatus: { cannotReconcile: { audit: ['permission denied'] } } }, + }) + const d = buildCNPGFleet(resp({ clusters: [c] })).rows[0].declarations + expect(d.total).toBe(2) + expect(d.failed).toBe(1) + expect(d.summary.tone).toBe('degraded') + }) + + it('treats declared managed roles without status as pending, not reconciled', () => { + const c = cluster('pg-a', 'db', { spec: { managed: { roles: [{ name: 'app' }, { name: 'audit' }] } } }) + const d = buildCNPGFleet(resp({ clusters: [c] })).rows[0].declarations + expect(d.pending).toBe(2) + expect(d.summary.text).toBe('2 of 2 pending') + }) + + it('does not claim "no schedule" when ScheduledBackups are unreadable', () => { + const fleet = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { scheduledBackups: { state: 'denied' } } })) + expect(fleet.rows[0].protection.summary.text).not.toBe('No backup destination or schedule') + }) + + it('says recovery is only declared until the restored cluster has a ready instance', () => { + const src = cluster('pg-a', 'db') + const restoring = cluster('pg-a-restore', 'db', { + spec: { bootstrap: { recovery: { backup: { name: 'nightly-1' } } } }, + status: { readyInstances: 0, currentPrimary: undefined }, + }) + const backups = [{ apiVersion: G, kind: 'Backup', metadata: { name: 'nightly-1', namespace: 'db' }, spec: { cluster: { name: 'pg-a' } }, status: { phase: 'completed' } }] + const a = buildCNPGFleet(resp({ clusters: [src, restoring], backups })).rows.find((r) => r.name === 'pg-a')! + expect(a.protection.restoreValidation.text).toBe('Recovery declared in pg-a-restore') + expect(a.protection.restoreValidation.tone).toBe('unknown') + }) + + it('does not say no restore was recorded when the Backup it names is unreadable', () => { + const src = cluster('pg-a', 'db') + const restoring = cluster('pg-a-restore', 'db', { spec: { bootstrap: { recovery: { backup: { name: 'nightly-1' } } } } }) + const a = buildCNPGFleet(resp({ clusters: [src, restoring] }, { coverage: { backups: { state: 'denied' } } })).rows.find((r) => r.name === 'pg-a')! + expect(a.protection.restoreValidation.text).toBe('Unknown: no access to Backups') + }) + + it('does not attribute a restore by backup-name prefix alone', () => { + const src = cluster('pg', 'db') + const other = cluster('pg-orders-restore', 'db', { spec: { bootstrap: { recovery: { backup: { name: 'pg-orders-backup' } } } } }) + const backups = [{ apiVersion: G, kind: 'Backup', metadata: { name: 'pg-orders-backup', namespace: 'db' }, spec: { cluster: { name: 'pg-orders' } }, status: { phase: 'completed' } }] + const row = buildCNPGFleet(resp({ clusters: [src, other], backups })).rows.find((r) => r.name === 'pg')! + expect(row.protection.restoreValidation.text).toBe('None recorded') + }) + + it('marks Poolers unknown when they are not readable', () => { + const fleet = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { poolers: { state: 'denied' } } })) + expect(fleet.rows[0].poolersKnown).toBe(false) + }) + + it('attributes instance Pod issues to their cluster', () => { + const fleet = buildCNPGFleet( + resp({ clusters: [cluster('pg-a', 'db')], pods: [pod('pg-a-2', 'db', 'pg-a', 'replica', false)] }, { + issues: [{ id: 'p1', severity: 'critical', kind: 'Pod', namespace: 'db', name: 'pg-a-2', reason: 'CrashLoopBackOff', message: 'Back-off restarting failed container' }], + }), + ) + expect(fleet.rows[0].attention).toBe(true) + expect(fleet.rows[0].categories.has('availability')).toBe(true) + }) + + it('reads partial coverage by allowed namespaces and treats unnamed partial coverage as unknown', () => { + const named = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'partial', allowedNamespaces: ['db'] } } })) + expect(named.rows[0].protection.lastSuccessfulBackup.text).toBe('None observed') + const unnamed = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'partial' } } })) + expect(unnamed.rows[0].protection.lastSuccessfulBackup.text).toBe('No access to Backups') + }) + + it('says no access instead of "no replica pods" when Pods are unreadable', () => { + const fleet = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { pods: { state: 'denied' } } })) + expect(fleet.rows[0].replication.text).toBe('No access to Pods') + }) + + it('does not report the recovery window or last backup as absent when ObjectStores are unreadable', () => { + const c = cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } }) + const p = buildCNPGFleet(resp({ clusters: [c] }, { coverage: { objectStores: { state: 'denied' } } })).rows[0].protection + expect(p.recoveryWindow.text).toBe('No access to ObjectStores') + expect(p.lastSuccessfulBackup.text).toBe('No access to ObjectStores') + }) + + it('ignores recovery sources from other plugins that reuse the barman parameter names', () => { + const src = cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } }) + const other = cluster('pg-x', 'db', { + spec: { + bootstrap: { recovery: { source: 'origin' } }, + externalClusters: [{ name: 'origin', plugin: { name: 'some-other-plugin', parameters: { barmanObjectName: 'store', serverName: 'pg-a' } } }], + }, + }) + const row = buildCNPGFleet(resp({ clusters: [src, other] })).rows.find((r) => r.name === 'pg-a')! + expect(row.protection.restoreValidation.text).toBe('None recorded') + }) + + it('words unreadable ObjectStores by coverage state', () => { + const c = cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } }) + const p = buildCNPGFleet(resp({ clusters: [c] }, { coverage: { objectStores: { state: 'syncing' } } })).rows[0].protection + expect(p.lastSuccessfulBackup.text).toBe('Loading…') + expect(p.recoveryWindow.text).toBe('Loading…') + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/workspace.ts b/packages/k8s-ui/src/components/cnpg/workspace.ts new file mode 100644 index 0000000000..eb8b1a95c7 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/workspace.ts @@ -0,0 +1,700 @@ +// CloudNativePG workspace model: pure derivations over the /api/cnpg/workspace +// payload. Every fact here is something the cluster actually reports; when it +// does not report something the value is "unknown", never zero or healthy. + +import type { HealthLevel } from '../resources/resource-utils' +import { + CNPG_BARMAN_PLUGIN_NAME, + getCNPGClusterBackupConfig, + getCNPGClusterBarmanPlugin, + getCNPGClusterImageTag, + getCNPGClusterStatus, + getCNPGObjectStoreRecoveryWindows, + isApiGroup, +} from '../resources/resource-utils-cnpg' + +export const CNPG_WORKSPACE_KEYS = [ + 'clusters', + 'backups', + 'scheduledBackups', + 'poolers', + 'databases', + 'publications', + 'subscriptions', + 'imageCatalogs', + 'clusterImageCatalogs', + 'objectStores', + 'pods', +] as const + +export type CNPGWorkspaceKey = (typeof CNPG_WORKSPACE_KEYS)[number] + +export type CNPGCoverageState = 'full' | 'partial' | 'denied' | 'notInstalled' | 'syncing' | 'error' + +export interface CNPGKindCoverage { + state: CNPGCoverageState + /** Denied namespaces, named only when the caller supplied the candidate list. */ + deniedNamespaces?: string[] + /** For partial coverage: the namespaces that were read. */ + allowedNamespaces?: string[] +} + +export interface CNPGWorkspaceIssue { + id: string + severity: 'critical' | 'warning' + category?: string + kind: string + group?: string + namespace?: string + name: string + reason: string + message?: string + cause?: string + action?: string + first_seen?: string +} + +export interface CNPGAuditFinding { + checkId: string + severity: string + kind: string + group?: string + namespace?: string + name: string + message: string +} + +export interface CNPGWorkspaceResponse { + installed: boolean + context: string + namespaces: string[] | null + coverage: Partial> + objects: Partial> + issues: CNPGWorkspaceIssue[] + audit: CNPGAuditFinding[] + backupsOmitted: number +} + +export const CNPG_KIND_BY_KEY: Record = { + clusters: { kind: 'Cluster', group: 'postgresql.cnpg.io', plural: 'clusters' }, + backups: { kind: 'Backup', group: 'postgresql.cnpg.io', plural: 'backups' }, + scheduledBackups: { kind: 'ScheduledBackup', group: 'postgresql.cnpg.io', plural: 'scheduledbackups' }, + poolers: { kind: 'Pooler', group: 'postgresql.cnpg.io', plural: 'poolers' }, + databases: { kind: 'Database', group: 'postgresql.cnpg.io', plural: 'databases' }, + publications: { kind: 'Publication', group: 'postgresql.cnpg.io', plural: 'publications' }, + subscriptions: { kind: 'Subscription', group: 'postgresql.cnpg.io', plural: 'subscriptions' }, + imageCatalogs: { kind: 'ImageCatalog', group: 'postgresql.cnpg.io', plural: 'imagecatalogs' }, + clusterImageCatalogs: { kind: 'ClusterImageCatalog', group: 'postgresql.cnpg.io', plural: 'clusterimagecatalogs' }, + objectStores: { kind: 'ObjectStore', group: 'barmancloud.cnpg.io', plural: 'objectstores' }, + pods: { kind: 'Pod', group: '', plural: 'pods' }, +} + +export function isCNPGWorkspaceKind(kind: string, group: string | undefined): boolean { + return Object.values(CNPG_KIND_BY_KEY).some((k) => k.group !== '' && k.group === (group ?? '') && k.kind === kind) +} + +/** The value is observed, derived, or not available from the cluster. */ +export type CNPGFactTone = HealthLevel + +export interface CNPGFact { + text: string + tone: CNPGFactTone + /** Where the value comes from, shown next to it so claims carry their source. */ + source?: string + /** A timestamp the text refers to; the UI renders it as an age. */ + at?: string +} + +export type CNPGProblemCategory = 'availability' | 'protection' | 'declarations' | 'pooling' + +export const CNPG_PROBLEM_CATEGORIES: { id: CNPGProblemCategory; label: string }[] = [ + { id: 'availability', label: 'Availability' }, + { id: 'protection', label: 'Protection' }, + { id: 'declarations', label: 'Declarations' }, + { id: 'pooling', label: 'Pooling' }, +] + +export interface CNPGProblem { + /** Stable identity for keys. */ + id: string + severity: 'critical' | 'warning' | 'posture' + category: CNPGProblemCategory + title: string + detail?: string + /** The object the evidence is about (may be the Cluster or a child object). */ + subject: { kind: string; group: string; namespace: string; name: string } + source: 'issue' | 'audit' +} + +export interface CNPGInstance { + name: string + role: 'primary' | 'replica' | 'unknown' + ready: boolean | null + node?: string + zone?: string +} + +export interface CNPGProtectionFacts { + schedule: CNPGFact & { names: string[] } + destination: CNPGFact & { + method: 'plugin' | 'barmanObjectStore' | 'volumeSnapshot' | 'none' + objectStore?: string + } + lastSuccessfulBackup: CNPGFact + walArchiving: CNPGFact + recoveryWindow: CNPGFact & { from?: string } + restoreValidation: CNPGFact & { restoredInto?: { namespace: string; name: string } } +} + +export interface CNPGFleetRow { + key: string + namespace: string + name: string + cluster: any + controllerStatus: { text: string; level: HealthLevel } + instances: { ready: number | null; desired: number | null } + pods: CNPGInstance[] + replicaCluster: { source?: string } | null + hibernated: boolean + pgVersion: string | null + catalog: { kind: string; name: string } | null + replication: CNPGFact + protection: CNPGProtectionFacts & { summary: CNPGFact } + declarations: { summary: CNPGFact; total: number; failed: number; pending: number } + poolers: string[] + /** False when Poolers are not readable in this cluster's namespace, so an empty list means unknown. */ + poolersKnown: boolean + problems: CNPGProblem[] + /** Any issue at warning or worse. Posture findings alone do not need attention. */ + attention: boolean + categories: Set + /** GitOps owner recorded on the Cluster, when it carries the standard labels. */ + gitops: CNPGGitOpsSource | null +} + +export interface CNPGFleet { + rows: CNPGFleetRow[] + attentionCount: number + categoryCounts: Record + /** Kinds whose coverage is not complete, so counts built on them are lower bounds. */ + incompleteKinds: CNPGWorkspaceKey[] +} + +const PROTECTION_ISSUE_REASONS = new Set([ + 'CNPGWALArchivingFailing', + 'CNPGLastBackupFailed', + 'CNPGBackupFailed', + 'CNPGScheduledBackupMissed', +]) + +export function cnpgIssueCategory(issue: Pick): CNPGProblemCategory { + if (PROTECTION_ISSUE_REASONS.has(issue.reason)) return 'protection' + switch (issue.kind) { + case 'Backup': + case 'ScheduledBackup': + case 'ObjectStore': + return 'protection' + case 'Database': + case 'Publication': + case 'Subscription': + return 'declarations' + case 'Pooler': + return 'pooling' + default: + return 'availability' + } +} + +function key(ns: string | undefined, name: string): string { + return `${ns ?? ''}/${name}` +} + +function specClusterName(obj: any): string | undefined { + const n = obj?.spec?.cluster?.name + return typeof n === 'string' && n ? n : undefined +} + +function coverageOf(resp: CNPGWorkspaceResponse, k: CNPGWorkspaceKey): CNPGKindCoverage { + return resp.coverage?.[k] ?? { state: 'notInstalled' } +} + +/** Coverage is usable in a namespace when that namespace's objects were read. */ +export function coverageReadable(cov: CNPGKindCoverage, namespace?: string): boolean { + if (cov.state === 'full') return true + if (cov.state === 'partial') { + if (!namespace) return false + if (cov.allowedNamespaces) return cov.allowedNamespaces.includes(namespace) + if (cov.deniedNamespaces) return !cov.deniedNamespaces.includes(namespace) + return false + } + return false +} + +function coverageUnavailableText(cov: CNPGKindCoverage, what: string): string { + switch (cov.state) { + case 'denied': + case 'partial': + return `No access to ${what}` + case 'syncing': + return 'Loading…' + case 'error': + return `Could not read ${what}` + default: + return 'Not installed' + } +} + +function timeOf(obj: any): number { + const t = obj?.status?.stoppedAt || obj?.status?.startedAt || obj?.metadata?.creationTimestamp + const ms = t ? Date.parse(t) : NaN + return Number.isFinite(ms) ? ms : 0 +} + +function podRole(pod: any, cluster: any): CNPGInstance['role'] { + const role = pod?.metadata?.labels?.['cnpg.io/instanceRole'] ?? pod?.metadata?.labels?.role + if (role === 'primary') return 'primary' + if (role === 'replica') return 'replica' + const primary = cluster?.status?.currentPrimary + if (primary && pod?.metadata?.name === primary) return 'primary' + return 'unknown' +} + +function podReady(pod: any): boolean | null { + const conds = pod?.status?.conditions + if (!Array.isArray(conds)) return null + const ready = conds.find((c: any) => c?.type === 'Ready') + if (!ready) return null + return ready.status === 'True' +} + +export type CNPGGitOpsSource = { tool: 'argocd' | 'flux'; name: string; namespace?: string } + +/** The GitOps owner recorded on an object's standard Argo CD / Flux labels. */ +export function cnpgGitOpsSource(obj: any): CNPGGitOpsSource | null { + const labels = obj?.metadata?.labels ?? {} + const annotations = obj?.metadata?.annotations ?? {} + const argo = labels['argocd.argoproj.io/instance'] + if (argo) return { tool: 'argocd', name: argo } + const tracking = annotations['argocd.argoproj.io/tracking-id'] + if (typeof tracking === 'string' && tracking.includes(':')) { + return { tool: 'argocd', name: tracking.split(':')[0] } + } + const fluxName = labels['kustomize.toolkit.fluxcd.io/name'] || labels['helm.toolkit.fluxcd.io/name'] + if (fluxName) { + return { + tool: 'flux', + name: fluxName, + namespace: labels['kustomize.toolkit.fluxcd.io/namespace'] || labels['helm.toolkit.fluxcd.io/namespace'], + } + } + return null +} + +function scheduleFact( + cluster: any, + schedules: any[], + cov: CNPGKindCoverage, +): CNPGProtectionFacts['schedule'] { + const ns = cluster.metadata?.namespace + if (!coverageReadable(cov, ns)) { + return { text: coverageUnavailableText(cov, 'ScheduledBackups'), tone: 'unknown', names: [] } + } + const mine = schedules.filter((s) => s.metadata?.namespace === ns && specClusterName(s) === cluster.metadata?.name) + if (mine.length === 0) return { text: 'No declarative schedule', tone: 'neutral', names: [] } + const active = mine.filter((s) => s.spec?.suspend !== true) + const names = mine.map((s) => s.metadata?.name).filter(Boolean) + if (active.length === 0) { + return { text: mine.length === 1 ? 'Schedule suspended' : 'All schedules suspended', tone: 'degraded', names } + } + const cron = active[0]?.spec?.schedule + return { + text: active.length === 1 ? (cron ? `Scheduled · ${cron}` : 'Scheduled') : `${active.length} schedules`, + tone: 'healthy', + names, + } +} + +function destinationFact(cluster: any): CNPGProtectionFacts['destination'] { + const plugin = getCNPGClusterBarmanPlugin(cluster) + if (plugin?.barmanObjectName) { + return { + text: `ObjectStore ${plugin.barmanObjectName}`, + tone: 'neutral', + method: 'plugin', + objectStore: plugin.barmanObjectName, + } + } + const cfg = getCNPGClusterBackupConfig(cluster) + if (cfg.destinationPath) { + return { text: cfg.destinationPath, tone: 'neutral', method: 'barmanObjectStore' } + } + if (cluster?.spec?.backup?.volumeSnapshot) { + return { text: 'Volume snapshots', tone: 'neutral', method: 'volumeSnapshot' } + } + return { text: 'No destination configured', tone: 'neutral', method: 'none' } +} + +function recoveryWindowFor(cluster: any, stores: any[]): { from?: string; lastSuccess?: string; store?: string } | null { + const plugin = getCNPGClusterBarmanPlugin(cluster) + if (!plugin?.barmanObjectName) return null + const store = stores.find( + (s) => s.metadata?.namespace === cluster.metadata?.namespace && s.metadata?.name === plugin.barmanObjectName, + ) + if (!store) return null + const server = plugin.serverName || cluster.metadata?.name + const w = getCNPGObjectStoreRecoveryWindows(store).find((x) => x.server === server) + if (!w) return { store: store.metadata?.name } + return { + from: w.firstRecoverabilityPoint, + lastSuccess: w.lastSuccessfulBackupTime, + store: store.metadata?.name, + } +} + +function lastBackupFact( + cluster: any, + backups: any[], + backupsCov: CNPGKindCoverage, + window: ReturnType, + storesUnreadable: CNPGKindCoverage | null, +): CNPGProtectionFacts['lastSuccessfulBackup'] { + const ns = cluster.metadata?.namespace + const name = cluster.metadata?.name + const candidates: { at: string; source: string }[] = [] + if (coverageReadable(backupsCov, ns)) { + const completed = backups + .filter( + (b) => + b.metadata?.namespace === ns && + specClusterName(b) === name && + isApiGroup(b.apiVersion, 'postgresql.cnpg.io') && + b.status?.phase === 'completed', + ) + .sort((a, b) => timeOf(b) - timeOf(a))[0] + const at = completed?.status?.stoppedAt || completed?.status?.startedAt + if (completed && at) candidates.push({ at, source: `Backup ${completed.metadata?.name}` }) + } + if (window?.lastSuccess) candidates.push({ at: window.lastSuccess, source: `ObjectStore ${window.store} status` }) + const cfg = getCNPGClusterBackupConfig(cluster) + if (!cfg.plugin && cfg.lastSuccessfulBackup) candidates.push({ at: cfg.lastSuccessfulBackup, source: 'Cluster status' }) + if (candidates.length === 0) { + if (!coverageReadable(backupsCov, ns)) { + return { text: coverageUnavailableText(backupsCov, 'Backups'), tone: 'unknown' } + } + if (storesUnreadable) return { text: coverageUnavailableText(storesUnreadable, 'ObjectStores'), tone: 'unknown' } + return { text: 'None observed', tone: 'unknown' } + } + const best = candidates.reduce((a, b) => (Date.parse(a.at) >= Date.parse(b.at) ? a : b)) + return { text: 'Completed', tone: 'healthy', at: best.at, source: best.source } +} + +function walFact(cluster: any): CNPGFact { + const conds = cluster?.status?.conditions + const c = Array.isArray(conds) ? conds.find((x: any) => x?.type === 'ContinuousArchiving') : null + if (!c) return { text: 'Not reported', tone: 'unknown', source: 'Cluster status' } + if (c.status === 'True') return { text: 'Archiving', tone: 'healthy', source: 'ContinuousArchiving condition' } + if (c.status === 'False') { + return { text: c.message ? `Failing · ${c.message}` : 'Failing', tone: 'unhealthy', source: 'ContinuousArchiving condition' } + } + return { text: 'Unknown', tone: 'unknown', source: 'ContinuousArchiving condition' } +} + +function restoreValidationFact( + cluster: any, + allClusters: any[], + backups: any[], + backupsReadable: boolean, +): CNPGProtectionFacts['restoreValidation'] { + const plugin = getCNPGClusterBarmanPlugin(cluster) + const server = plugin?.serverName || cluster.metadata?.name + const store = plugin?.barmanObjectName + const ns = cluster.metadata?.namespace + const name = cluster.metadata?.name + const restored = allClusters.find((c) => { + if (c === cluster || c.metadata?.namespace !== ns) return false + const recovery = c.spec?.bootstrap?.recovery + if (!recovery) return false + const sourceName = recovery.source + if (sourceName) { + const ext = (c.spec?.externalClusters ?? []).find((e: any) => e?.name === sourceName) + const params = ext?.plugin?.name === CNPG_BARMAN_PLUGIN_NAME ? ext.plugin.parameters : undefined + if (store && params?.barmanObjectName === store && (params?.serverName || sourceName) === server) return true + } + const backupName = recovery.backup?.name + if (!backupName) return false + const backup = backups.find((b) => b.metadata?.namespace === ns && b.metadata?.name === backupName) + return specClusterName(backup) === name + }) + if (!restored) { + // A recovery by Backup name is only attributable when that Backup could be read. + const unresolved = !backupsReadable && allClusters.some((c) => c !== cluster && c.metadata?.namespace === ns && c.spec?.bootstrap?.recovery?.backup?.name) + if (unresolved) return { text: 'Unknown: no access to Backups', tone: 'unknown', source: `A Cluster in ${ns} recovers from a Backup Radar cannot read` } + return { text: 'None recorded', tone: 'unknown', source: 'Kubernetes does not record restore tests' } + } + const rname = restored.metadata?.name + const ready = typeof restored.status?.readyInstances === 'number' && restored.status.readyInstances > 0 + if (!ready) { + return { + text: `Recovery declared in ${rname}`, + tone: 'unknown', + source: `Cluster ${rname} bootstraps from this cluster's backups but has no ready instance yet`, + restoredInto: { namespace: restored.metadata?.namespace, name: rname }, + } + } + return { + text: `Restored into ${rname}`, + tone: 'neutral', + source: `Cluster ${rname} bootstrapped from this cluster's backups and has ready instances · created ${restored.metadata?.creationTimestamp ?? 'unknown'}. This proves one recovery, not that today's backups restore.`, + restoredInto: { namespace: restored.metadata?.namespace, name: rname }, + } +} + +function protectionSummary(p: CNPGProtectionFacts): CNPGFact { + if (p.walArchiving.tone === 'unhealthy') return { text: 'WAL archiving failing', tone: 'unhealthy' } + if (p.destination.method === 'none' && p.schedule.names.length === 0 && p.schedule.tone !== 'unknown') { + return { text: 'No backup destination or schedule', tone: 'neutral' } + } + if (p.schedule.tone === 'degraded') return { text: p.schedule.text, tone: 'degraded' } + if (p.lastSuccessfulBackup.at) return { text: 'Last backup', tone: 'healthy', at: p.lastSuccessfulBackup.at } + return { text: p.lastSuccessfulBackup.text, tone: p.lastSuccessfulBackup.tone } +} + +function catalogRef(cluster: any): CNPGFleetRow['catalog'] { + const ref = cluster?.spec?.imageCatalogRef + if (!ref?.name) return null + return { kind: ref.kind || 'ImageCatalog', name: ref.name } +} + +function pgVersion(cluster: any): string | null { + const tag = getCNPGClusterImageTag(cluster) + if (tag && tag !== '-') { + const m = tag.match(/^(\d+(?:\.\d+)?)/) + if (m) return m[1] + } + const major = cluster?.spec?.imageCatalogRef?.major + return typeof major === 'number' ? String(major) : null +} + +function replicationFact(cluster: any, pods: CNPGInstance[], hibernated: boolean, podsCov: CNPGKindCoverage): CNPGFact { + if (hibernated) return { text: 'Hibernated', tone: 'neutral' } + const desired = cluster?.spec?.instances + if (desired === 1) return { text: 'Single instance', tone: 'neutral' } + if (!coverageReadable(podsCov, cluster?.metadata?.namespace)) { + return { text: coverageUnavailableText(podsCov, 'Pods'), tone: 'unknown' } + } + const replicas = pods.filter((p) => p.role === 'replica') + const readyReplicas = replicas.filter((p) => p.ready === true).length + if (replicas.length === 0) return { text: 'No replica pods observed', tone: 'unknown' } + return { + text: `${readyReplicas}/${replicas.length} replicas ready · lag unknown`, + tone: 'unknown', + source: 'Pod readiness does not show whether a replica is streaming', + } +} + +function problemsFor( + cluster: any, + issues: CNPGWorkspaceIssue[], + audit: CNPGAuditFinding[], + children: Map, +): CNPGProblem[] { + const ns = cluster.metadata?.namespace + const name = cluster.metadata?.name + const out: CNPGProblem[] = [] + for (const issue of issues) { + if ((issue.namespace ?? '') !== ns) continue + const isSelf = issue.kind === 'Cluster' && issue.name === name + const owner = children.get(`${issue.kind}/${ns}/${issue.name}`) + if (!isSelf && owner !== name) continue + out.push({ + id: `${issue.id}:${issue.kind}/${issue.name}`, + severity: issue.severity, + category: cnpgIssueCategory(issue), + title: issue.message || issue.reason, + detail: issue.cause || undefined, + subject: { kind: issue.kind, group: issue.group ?? '', namespace: ns, name: issue.name }, + source: 'issue', + }) + } + for (const f of audit) { + if (f.kind !== 'Cluster' || f.name !== name || (f.namespace ?? '') !== ns) continue + out.push({ + id: `audit:${f.checkId}:${ns}/${name}`, + severity: 'posture', + category: 'protection', + title: f.checkId === 'cnpgNoDeclarativeBackup' ? 'No declarative backup schedule' : f.message, + detail: f.message, + subject: { kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: ns, name }, + source: 'audit', + }) + } + const rank = { critical: 0, warning: 1, posture: 2 } as const + return out.sort((a, b) => rank[a.severity] - rank[b.severity] || a.title.localeCompare(b.title)) +} + +/** Index "Kind/ns/name" → owning cluster name, from each child's spec.cluster.name. */ +function childIndex(resp: CNPGWorkspaceResponse): Map { + const idx = new Map() + const add = (kind: string, list: any[] | undefined) => { + for (const o of list ?? []) { + const c = specClusterName(o) + if (c) idx.set(`${kind}/${o.metadata?.namespace}/${o.metadata?.name}`, c) + } + } + add('Backup', resp.objects.backups) + add('ScheduledBackup', resp.objects.scheduledBackups) + add('Pooler', resp.objects.poolers) + add('Database', resp.objects.databases) + add('Publication', resp.objects.publications) + add('Subscription', resp.objects.subscriptions) + for (const p of resp.objects.pods ?? []) { + const c = p?.metadata?.labels?.['cnpg.io/cluster'] + if (c) idx.set(`Pod/${p.metadata?.namespace}/${p.metadata?.name}`, c) + } + return idx +} + +function declarationsFor(cluster: any, resp: CNPGWorkspaceResponse): CNPGFleetRow['declarations'] { + const ns = cluster.metadata?.namespace + const name = cluster.metadata?.name + const lists: [CNPGWorkspaceKey, any[]][] = [ + ['databases', resp.objects.databases ?? []], + ['publications', resp.objects.publications ?? []], + ['subscriptions', resp.objects.subscriptions ?? []], + ] + let total = 0 + let failed = 0 + let pending = 0 + let unreadable = false + for (const [k, list] of lists) { + if (!coverageReadable(coverageOf(resp, k), ns)) { + if (coverageOf(resp, k).state !== 'notInstalled') unreadable = true + continue + } + for (const o of list) { + if (o.metadata?.namespace !== ns || specClusterName(o) !== name) continue + total++ + if (o.status?.applied === false) failed++ + else if (o.status?.applied !== true) pending++ + } + } + const roleStatus = cluster?.status?.managedRolesStatus + const reconciledRoles = new Set(roleStatus?.byStatus?.reconciled ?? []) + const failedRoles = new Set(Object.keys(roleStatus?.cannotReconcile ?? {})) + const declaredRoles: any[] = Array.isArray(cluster?.spec?.managed?.roles) ? cluster.spec.managed.roles : [] + for (const r of declaredRoles) { + if (!r?.name) continue + total++ + if (failedRoles.has(r.name)) failed++ + else if (!reconciledRoles.has(r.name)) pending++ + } + let summary: CNPGFact + if (total === 0) { + summary = unreadable ? { text: 'No access to some declarations', tone: 'unknown' } : { text: 'None declared', tone: 'neutral' } + } else if (failed > 0) { + summary = { text: `${failed} of ${total} not reconciled`, tone: 'degraded' } + } else if (pending > 0) { + summary = { text: `${pending} of ${total} pending`, tone: 'unknown' } + } else { + summary = { text: `${total} reconciled`, tone: 'healthy' } + } + if (unreadable && total > 0) summary = { ...summary, source: 'Some declaration kinds are not readable' } + return { summary, total, failed, pending } +} + +export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { + const clusters = (resp.objects.clusters ?? []).filter((c) => isApiGroup(c?.apiVersion, 'postgresql.cnpg.io')) + const pods = resp.objects.pods ?? [] + const stores = resp.objects.objectStores ?? [] + const children = childIndex(resp) + const poolers = resp.objects.poolers ?? [] + + const rows: CNPGFleetRow[] = clusters.map((cluster) => { + const ns = cluster.metadata?.namespace ?? '' + const name = cluster.metadata?.name ?? '' + const status = getCNPGClusterStatus(cluster) + const hibernated = cluster?.metadata?.annotations?.['cnpg.io/hibernation'] === 'on' + const instancePods: CNPGInstance[] = pods + .filter((p) => p.metadata?.namespace === ns && p.metadata?.labels?.['cnpg.io/cluster'] === name) + .map((p) => ({ + name: p.metadata?.name, + role: podRole(p, cluster), + ready: podReady(p), + node: p.spec?.nodeName, + })) + .sort((a, b) => (a.role === 'primary' ? -1 : b.role === 'primary' ? 1 : a.name.localeCompare(b.name))) + const readyInstances = typeof cluster?.status?.readyInstances === 'number' ? cluster.status.readyInstances : null + const desired = typeof cluster?.spec?.instances === 'number' ? cluster.spec.instances : null + + const window = recoveryWindowFor(cluster, stores) + const wal = walFact(cluster) + const storesCov = coverageOf(resp, 'objectStores') + const storesUnreadable = !!getCNPGClusterBarmanPlugin(cluster)?.barmanObjectName && !coverageReadable(storesCov, ns) + const protection: CNPGProtectionFacts = { + schedule: scheduleFact(cluster, resp.objects.scheduledBackups ?? [], coverageOf(resp, 'scheduledBackups')), + destination: destinationFact(cluster), + lastSuccessfulBackup: lastBackupFact(cluster, resp.objects.backups ?? [], coverageOf(resp, 'backups'), window, storesUnreadable ? storesCov : null), + walArchiving: wal, + // The latest recoverable point follows WAL archiving, not the last base + // backup, and no status reports it; only failing archiving stops it. + recoveryWindow: window?.from + ? { + text: 'Recoverable window', + tone: wal.tone === 'unhealthy' ? 'degraded' : 'neutral', + from: window.from, + source: `ObjectStore ${window.store} status (earliest point)`, + } + : storesUnreadable + ? { text: coverageUnavailableText(storesCov, 'ObjectStores'), tone: 'unknown' } + : { text: 'Not reported', tone: 'unknown' }, + restoreValidation: restoreValidationFact(cluster, clusters, resp.objects.backups ?? [], coverageReadable(coverageOf(resp, 'backups'), ns)), + } + const problems = problemsFor(cluster, resp.issues ?? [], resp.audit ?? [], children) + const categories = new Set( + problems.filter((p) => p.severity !== 'posture').map((p) => p.category), + ) + const replica = cluster?.spec?.replica?.enabled ? { source: cluster.spec.replica.source } : null + + return { + key: key(ns, name), + namespace: ns, + name, + cluster, + controllerStatus: { text: status.text, level: status.level }, + instances: { ready: readyInstances, desired }, + pods: instancePods, + replicaCluster: replica, + hibernated, + pgVersion: pgVersion(cluster), + catalog: catalogRef(cluster), + replication: replicationFact(cluster, instancePods, hibernated, coverageOf(resp, 'pods')), + protection: { ...protection, summary: protectionSummary(protection) }, + declarations: declarationsFor(cluster, resp), + poolers: poolers + .filter((p) => p.metadata?.namespace === ns && specClusterName(p) === name) + .map((p) => p.metadata?.name), + poolersKnown: coverageReadable(coverageOf(resp, 'poolers'), ns), + problems, + attention: problems.some((p) => p.severity !== 'posture'), + categories, + gitops: cnpgGitOpsSource(cluster), + } + }) + + rows.sort((a, b) => Number(b.attention) - Number(a.attention) || a.namespace.localeCompare(b.namespace) || a.name.localeCompare(b.name)) + + const categoryCounts = { availability: 0, protection: 0, declarations: 0, pooling: 0 } as Record + for (const r of rows) for (const c of r.categories) categoryCounts[c]++ + + const incompleteKinds = CNPG_WORKSPACE_KEYS.filter((k) => { + const s = coverageOf(resp, k).state + return s === 'partial' || s === 'denied' || s === 'syncing' || s === 'error' + }) + + return { + rows, + attentionCount: rows.filter((r) => r.attention).length, + categoryCounts, + incompleteKinds, + } +} diff --git a/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx b/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx index 44dce73e23..7429886c6f 100644 --- a/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx +++ b/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx @@ -63,9 +63,15 @@ export interface WorkloadLogsViewerProps { * re-armed. Requires `createStream`. Default: false. */ autoStream?: boolean + /** Pods selected when the pod list first loads; all pods when empty or none match. */ + initialPods?: string[] } -export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownload, forceDark, defaultDark, autoStream = false }: WorkloadLogsViewerProps) { +export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownload, forceDark, defaultDark, autoStream = false, initialPods }: WorkloadLogsViewerProps) { + const initialSelection = (names: string[]) => { + const wanted = names.filter((n) => initialPods?.includes(n)) + return new Set(wanted.length > 0 ? wanted : names) + } const [selectedContainer, setSelectedContainer] = useState('') const [pods, setPods] = useState([]) const [selectedPods, setSelectedPods] = useState>(new Set()) @@ -129,7 +135,9 @@ export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownl const previousPods = previousSnapshotPods.current const nextPods = resultPods.map(p => p.name) previousSnapshotPods.current = nextPods - setSelectedPods(selected => previousPods === null || previousPods.every(pod => selected.has(pod)) + setSelectedPods(selected => previousPods === null + ? initialSelection(nextPods) + : previousPods.every(pod => selected.has(pod)) ? new Set(nextPods) : new Set(nextPods.filter(pod => selected.has(pod)))) @@ -194,7 +202,7 @@ export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownl setEmptyMessage(data.emptyMessage || null) setEmptyCommand(data.command || null) setSelectedPods(prev => ( - prev.size === 0 ? new Set(nextPods.map((p: WorkloadPodInfo) => p.name)) : prev + prev.size === 0 ? initialSelection(nextPods.map((p: WorkloadPodInfo) => p.name)) : prev )) } }, @@ -205,6 +213,7 @@ export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownl content: data.content || '', container: data.container || '', pod: data.pod || '', + sourceLabel: data.sourceLabel, podColorIndex: podColorIndexRef.current.get(data.pod || ''), }) } diff --git a/packages/k8s-ui/src/components/logs/useLogBuffer.test.ts b/packages/k8s-ui/src/components/logs/useLogBuffer.test.ts new file mode 100644 index 0000000000..306de1b4eb --- /dev/null +++ b/packages/k8s-ui/src/components/logs/useLogBuffer.test.ts @@ -0,0 +1,17 @@ +import { describe, expect, it } from 'vitest' +import { detectLogLevel } from './useLogBuffer' + +describe('detectLogLevel with CloudNativePG records', () => { + it('uses the nested PostgreSQL severity over the instance manager level', () => { + const line = (sev: string) => JSON.stringify({ level: 'info', logger: 'postgres', msg: 'record', record: { error_severity: sev, message: 'x' } }) + expect(detectLogLevel(line('ERROR'))).toBe('error') + expect(detectLogLevel(line('FATAL'))).toBe('error') + expect(detectLogLevel(line('WARNING'))).toBe('warn') + expect(detectLogLevel(line('LOG'))).toBe('info') + expect(detectLogLevel(line('DEBUG'))).toBe('debug') + }) + it('ignores a record that is not a PostgreSQL one', () => { + expect(detectLogLevel(JSON.stringify({ level: 'error', msg: 'request failed', record: { error_severity: 20 } }))).toBe('error') + expect(detectLogLevel(JSON.stringify({ level: 'error', msg: 'x', record: { error_severity: 'minor' } }))).toBe('error') + }) +}) diff --git a/packages/k8s-ui/src/components/resources/ResourcesSidebar.test.tsx b/packages/k8s-ui/src/components/resources/ResourcesSidebar.test.tsx index 39c0c9f465..f76df8d7f1 100644 --- a/packages/k8s-ui/src/components/resources/ResourcesSidebar.test.tsx +++ b/packages/k8s-ui/src/components/resources/ResourcesSidebar.test.tsx @@ -105,3 +105,79 @@ describe('ResourcesSidebar count visibility', () => { expect(html).toContain('HorizontalPodAutoscaler') }) }) + +describe('ResourcesSidebar category workspaces', () => { + const cnpgCluster: APIResource = { + group: 'postgresql.cnpg.io', + version: 'v1', + kind: 'Cluster', + name: 'clusters', + namespaced: true, + isCrd: true, + verbs: ['list'], + } + const objectStore: APIResource = { + group: 'barmancloud.cnpg.io', + version: 'v1', + kind: 'ObjectStore', + name: 'objectstores', + namespaced: true, + isCrd: true, + verbs: ['list'], + } + + it('renders destinations above the kinds, with the active object nested and no kind selected', () => { + const html = renderToString( + {}} + apiResources={[cnpgCluster, objectStore]} + resourceCounts={{ 'postgresql.cnpg.io/Cluster': 2, 'barmancloud.cnpg.io/ObjectStore': 1 }} + categoryWorkspaces={{ + CloudNativePG: { + destinations: [ + { id: 'overview', label: 'Overview', count: 3, countTitle: '3 clusters need attention', active: true, child: { label: 'pg-orders' }, onSelect: () => {} }, + ], + defaultKindsCollapsed: true, + scopeNote: 'Counts for namespace payments', + }, + }} + /> + ) + expect(html).toContain('Workspace') + expect(html).toContain('Overview') + expect(html).toContain('pg-orders') + expect(html).toContain('Counts for namespace payments') + expect(html).toContain('Resource kinds') + expect(html).toMatch(/aria-expanded="false"[^>]*>(?:(?!<\/button>).)*Resource kinds/) + expect(html).not.toContain('selection-strong selection-text">Pod') + }) + + it('keeps a workspace category visible when it has no resources', () => { + const html = renderToString( + {}} + apiResources={[cnpgCluster]} + resourceCounts={{ 'postgresql.cnpg.io/Cluster': 0 }} + categoryWorkspaces={{ CloudNativePG: { destinations: [{ id: 'overview', label: 'Overview', onSelect: () => {} }] } }} + /> + ) + expect(html).toContain('CloudNativePG') + expect(html).toContain('Overview') + }) + + it('labels API groups when a workspace category spans several', () => { + const html = renderToString( + {}} + apiResources={[cnpgCluster, objectStore]} + resourceCounts={{ 'postgresql.cnpg.io/Cluster': 2, 'barmancloud.cnpg.io/ObjectStore': 1 }} + categoryWorkspaces={{ CloudNativePG: { destinations: [{ id: 'overview', label: 'Overview', onSelect: () => {} }], defaultKindsCollapsed: true } }} + /> + ) + expect(html).toContain('postgresql.cnpg.io') + expect(html).toContain('barmancloud.cnpg.io') + }) +}) diff --git a/packages/k8s-ui/src/components/resources/ResourcesSidebar.tsx b/packages/k8s-ui/src/components/resources/ResourcesSidebar.tsx index 792a03ee8b..c06bc32e2b 100644 --- a/packages/k8s-ui/src/components/resources/ResourcesSidebar.tsx +++ b/packages/k8s-ui/src/components/resources/ResourcesSidebar.tsx @@ -1,4 +1,4 @@ -import { useState, useMemo, useEffect, useRef, useCallback, useId, forwardRef } from 'react' +import { useState, useMemo, useEffect, useRef, useCallback, useId, forwardRef, type ComponentType } from 'react' import { Search, Eye, @@ -29,6 +29,28 @@ export interface PinnedItem { group: string } +/** A task destination shown inside a category, above its exact kinds. */ +export interface SidebarCategoryDestination { + id: string + label: string + icon?: ComponentType<{ className?: string }> + /** Problem count. `undefined` renders no badge; `null` renders the unknown dash. */ + count?: number | null + countTitle?: string + active?: boolean + /** The object currently open under this destination, nested beneath it. */ + child?: { label: string; title?: string } + onSelect: () => void +} + +export interface SidebarCategoryWorkspace { + destinations: SidebarCategoryDestination[] + /** Kinds start collapsed under a "Resource kinds" disclosure until the user opens them. */ + defaultKindsCollapsed?: boolean + /** Explains what the destination counts are scoped to. */ + scopeNote?: string +} + export interface ResourcesSidebarProps { selectedKind: SelectedKindInfo | null onSelectedKindChange: (kind: SelectedKindInfo) => void @@ -48,10 +70,15 @@ export interface ResourcesSidebarProps { /** Called when a kind is selected via keyboard (Enter in the filter). Parent uses this * to move focus to the next UI level (e.g., the table search input). */ onKindNavigated?: () => void + /** Task destinations keyed by category name (e.g. "CloudNativePG"). A category + * with a workspace stays visible even when it has no resources. */ + categoryWorkspaces?: Record } // Persisted across remounts so collapsed categories survive tab switches let persistedExpandedCategories: Set | null = null +// Per-category "Resource kinds" disclosure, once the user has toggled it. +const persistedKindsOpen = new Map() const COUNT_UNAVAILABLE_MESSAGE = 'Count unavailable. Open to view resources.' // Fallback resource types when API resources aren't loaded yet @@ -102,10 +129,11 @@ interface ResourceTypeButtonProps { isPinned?: boolean onTogglePin?: () => void onClick: () => void + indent?: boolean } const ResourceTypeButton = forwardRef( - function ResourceTypeButton({ resource, count, isSelected, isHighlighted, isForbidden: forbidden, isPinned, onTogglePin, onClick }, ref) { + function ResourceTypeButton({ resource, count, isSelected, isHighlighted, isForbidden: forbidden, isPinned, onTogglePin, onClick, indent }, ref) { const Icon = getResourceIcon(resource.kind, resource.group) return ( + {d.child && ( +
+ {d.child.label} +
+ )} + + ) + })} + {workspace.scopeNote && ( +
{workspace.scopeNote}
+ )} + + + ) +} diff --git a/packages/k8s-ui/src/components/resources/ResourcesView.tsx b/packages/k8s-ui/src/components/resources/ResourcesView.tsx index b4cfafed10..a6f4a8027b 100644 --- a/packages/k8s-ui/src/components/resources/ResourcesView.tsx +++ b/packages/k8s-ui/src/components/resources/ResourcesView.tsx @@ -205,7 +205,7 @@ import { CalicoInfraCell, CalicoPolicyCell } from './renderers/calico-cells' import { isCalicoPolicyResource, isCoreNetworkPolicyKind } from './resource-utils-calico' import { useRegisterShortcut, useRegisterShortcuts } from '../../hooks/useKeyboardShortcuts' import { ResourcesSidebar } from './ResourcesSidebar' -import type { SelectedKindInfo } from './ResourcesSidebar' +import type { SelectedKindInfo, SidebarCategoryWorkspace } from './ResourcesSidebar' import { CompareTray, togglePick, pickIndex, refToParam, SIDE_TONES, type CompareTrayPick, type NamespacedRef } from '../compare' import { ConfirmDialog } from '../ui/ConfirmDialog' @@ -3411,6 +3411,8 @@ interface ResourcesViewProps { onSelectedKindChange?: (kind: { name: string; kind: string; group: string }) => void /** When true, the sidebar is not rendered. Useful when a standalone ResourcesSidebar is used externally. */ hideSidebar?: boolean + /** Task destinations rendered inside sidebar categories (see ResourcesSidebar). */ + sidebarCategoryWorkspaces?: Record /** Callback when the [+] create button is clicked. Receives the currently selected kind info. */ onCreateResource?: (kind: { name: string; kind: string; group: string } | null) => void /** Default kind when the URL does not include one. */ @@ -3653,6 +3655,7 @@ export function ResourcesView({ onOpenWorkloadLogs, onSelectedKindChange, hideSidebar = false, + sidebarCategoryWorkspaces, onCreateResource, defaultKind = DEFAULT_KIND_INFO, extraLeadingColumns, @@ -5867,6 +5870,7 @@ export function ResourcesView({ pinned={pinned} togglePin={togglePin} isPinned={isPinned} + categoryWorkspaces={sidebarCategoryWorkspaces} onKindNavigated={() => { // After selecting a kind via keyboard, move focus to the table search // so the user can immediately filter within the selected kind. diff --git a/packages/k8s-ui/src/components/resources/index.ts b/packages/k8s-ui/src/components/resources/index.ts index f872236ff1..ae7ff0a224 100644 --- a/packages/k8s-ui/src/components/resources/index.ts +++ b/packages/k8s-ui/src/components/resources/index.ts @@ -54,7 +54,7 @@ export * from './resource-utils-velero' export { ResourcesView, ResourcesViewDataContext, hasCuratedColumns } from './ResourcesView' export type { ResourceQueryResult, ExtraColumn, LargeListGuardState } from './ResourcesView' export { ResourcesSidebar } from './ResourcesSidebar' -export type { ResourcesSidebarProps, SelectedKindInfo, PinnedItem } from './ResourcesSidebar' +export type { ResourcesSidebarProps, SelectedKindInfo, PinnedItem, SidebarCategoryDestination, SidebarCategoryWorkspace } from './ResourcesSidebar' export { sanitizePrinterTable, printerTableKey, diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx index 4dee9b9a9e..40e78cb409 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx @@ -97,3 +97,28 @@ describe('a window that has stopped advancing', () => { expect(html).not.toContain('WAL archiving has stopped') }) }) + +describe('ObjectStore failing backups', () => { + const failing = { + ...store, + status: { serverRecoveryWindow: { 'pg-main': { firstRecoverabilityPoint: '2026-08-01T00:00:00Z', lastSuccessfulBackupTime: '2026-08-10T00:00:00Z', lastFailedBackupTime: '2026-08-11T00:00:00Z' } } }, + } + it('does not say recovery still reaches the newest WAL when archiving has stopped too', () => { + const html = renderToString() + expect(html).toContain('WAL archiving has stopped for pg-main') + expect(html).not.toContain('can still replay archived WAL') + expect(html).toContain('WAL archiving has stopped on the cluster behind this server') + }) + it('keeps recovery replaying archived WAL when only base backups fail', () => { + const html = renderToString() + expect(html).toContain('can still replay archived WAL written since') + expect(html).not.toContain('retention') + }) + it('claims nothing about recovery when no successful backup is recorded', () => { + const never = { ...store, status: { serverRecoveryWindow: { 'pg-main': { lastFailedBackupTime: '2026-08-11T00:00:00Z' } } } } + const html = renderToString() + expect(html).toContain('No successful base backup is recorded for pg-main') + expect(html).not.toContain('nothing to restore') + expect(html).not.toContain('replay') + }) +}) diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx index b9044db137..8b11cd328e 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx @@ -44,6 +44,8 @@ export function CNPGObjectStoreRenderer({ const credentialSecret = getCNPGObjectStoreCredentialSecret(data) const cfg = data?.spec?.configuration ?? {} const failing = windows.filter((w) => w.failingSinceLastSuccess) + const archiveStopped = failing.filter((w) => archivingFailing?.has(w.server)).map((w) => w.server) + const neverSucceeded = failing.filter((w) => !w.lastSuccessfulBackupTime).map((w) => w.server) return ( <> @@ -51,7 +53,13 @@ export function CNPGObjectStoreRenderer({ 0 + ? `No successful base backup is recorded for ${neverSucceeded.join(', ')}, so recoverability is not established.` + : archiveStopped.length > 0 + ? `A base backup failure is recorded after the last recorded success, and WAL archiving has stopped for ${archiveStopped.join(', ')}: nothing written since the last archived WAL can be recovered.` + : 'A base backup failure is recorded after the last recorded success. While WAL archiving works, recovery from that backup can still replay archived WAL written since.' + } /> )} @@ -60,7 +68,7 @@ export function CNPGObjectStoreRenderer({ // Configured but empty is NOT the same as healthy. Saying nothing here // would read as "backups are fine" on a store holding nothing.
- No server has reported a backup yet, so there is nothing to restore from this store. + No server has recorded a backup in this store's status, so recoverability is not established.
) : (
@@ -194,10 +202,10 @@ function RecoveryWindowRow({ ? 'Not advancing' : w.lastSuccessfulBackupTime ? 'Recoverable' - : 'No backups yet'} + : 'No backup recorded'}
- {stalled && !w.failingSinceLastSuccess && ( + {stalled && (
{/* The timestamps below are real and still describe the last backup that worked. What they no longer describe is a window still growing, and @@ -217,7 +225,7 @@ function RecoveryWindowRow({ {w.lastSuccessfulBackupTime ? ( } /> ) : ( - + )} {w.lastFailedBackupTime && ( } /> @@ -226,8 +234,9 @@ function RecoveryWindowRow({ {w.failingSinceLastSuccess && (
- Every backup since the last success has failed. The oldest restorable point will still age - out under the retention policy, so the window is shrinking from both ends. + {w.lastSuccessfulBackupTime + ? 'A backup failure is recorded after the last recorded success. Restoring from that success replays the WAL archived since, which takes longer the longer this lasts.' + : 'No successful backup is recorded for this server, so recoverability is not established.'}
)}
diff --git a/packages/k8s-ui/src/components/resources/resource-utils-cnpg-objectstore.test.ts b/packages/k8s-ui/src/components/resources/resource-utils-cnpg-objectstore.test.ts index caa1b7a29c..f9a69ed63c 100644 --- a/packages/k8s-ui/src/components/resources/resource-utils-cnpg-objectstore.test.ts +++ b/packages/k8s-ui/src/components/resources/resource-utils-cnpg-objectstore.test.ts @@ -77,7 +77,7 @@ describe('getCNPGObjectStoreStatus', () => { it('is unknown, not healthy, when no server has reported', () => { const s = getCNPGObjectStoreStatus(store({})) expect(s.level).toBe('unknown') - expect(s.text).toBe('No backups yet') + expect(s.text).toBe('No backup recorded') }) it('is unhealthy when any server is failing since its last success', () => { diff --git a/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts b/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts index 75bd8fa90a..ccae941c82 100644 --- a/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts +++ b/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts @@ -600,7 +600,7 @@ export interface CNPGObjectStoreRecoveryWindow { firstRecoverabilityPoint?: string lastSuccessfulBackupTime?: string lastFailedBackupTime?: string - /** A failure newer than the last success — the window has stopped advancing. */ + /** A failure newer than the last success, or a failure and no success at all. */ failingSinceLastSuccess: boolean } @@ -641,7 +641,7 @@ export function getCNPGObjectStoreRecoveryWindows(resource: any): CNPGObjectStor export function getCNPGObjectStoreStatus(resource: any): StatusBadge { const windows = getCNPGObjectStoreRecoveryWindows(resource) if (windows.length === 0) { - return { text: 'No backups yet', color: healthColors.unknown, level: 'unknown' } + return { text: 'No backup recorded', color: healthColors.unknown, level: 'unknown' } } if (windows.some((w) => w.failingSinceLastSuccess)) { return { text: 'Backups Failing', color: healthColors.unhealthy, level: 'unhealthy' } @@ -649,7 +649,7 @@ export function getCNPGObjectStoreStatus(resource: any): StatusBadge { // Every timestamp on a recovery window is optional, so a server can be listed // with no recovery point at all. The entry existing is not a backup existing. if (!windows.some((w) => w.firstRecoverabilityPoint || w.lastSuccessfulBackupTime)) { - return { text: 'No recovery point', color: healthColors.unknown, level: 'unknown' } + return { text: 'Recovery point not reported', color: healthColors.unknown, level: 'unknown' } } return { text: 'Recoverable', color: healthColors.healthy, level: 'healthy' } } diff --git a/packages/k8s-ui/src/components/resources/status-overclaim.test.ts b/packages/k8s-ui/src/components/resources/status-overclaim.test.ts index 407b2793df..bef747ccf7 100644 --- a/packages/k8s-ui/src/components/resources/status-overclaim.test.ts +++ b/packages/k8s-ui/src/components/resources/status-overclaim.test.ts @@ -141,7 +141,7 @@ describe('getCNPGObjectStoreStatus', () => { // Every timestamp on a RecoveryWindow is optional, so the server can be // listed while holding nothing restorable. expect(getCNPGObjectStoreStatus({ status: { serverRecoveryWindow: { pg: {} } } })) - .toMatchObject({ text: 'No recovery point', level: 'unknown' }) + .toMatchObject({ text: 'Recovery point not reported', level: 'unknown' }) }) it('reports recoverable once a real point exists', () => { diff --git a/packages/k8s-ui/src/components/workload/WorkloadView.tsx b/packages/k8s-ui/src/components/workload/WorkloadView.tsx index 8e6b83f729..c6e1f3c937 100644 --- a/packages/k8s-ui/src/components/workload/WorkloadView.tsx +++ b/packages/k8s-ui/src/components/workload/WorkloadView.tsx @@ -77,9 +77,19 @@ import { rolloutMayAdvanceAutomatically, type WorkloadRolloutActivity } from '.. import { WorkloadRolloutNotice } from './WorkloadRolloutNotice' import { isCoreBatchJob } from '../../utils/api-resources' -export type WorkloadTabType = 'overview' | 'topology' | 'timeline' | 'logs' | 'metrics' | 'reachability' | 'cost' | 'yaml' +export type WorkloadTabType = 'overview' | 'spec' | 'topology' | 'timeline' | 'logs' | 'metrics' | 'reachability' | 'cost' | 'yaml' type TabType = WorkloadTabType +/** A host-provided tab. `after` places it behind a built-in tab; `replaces` hides that built-in tab. */ +export interface WorkloadExtraTab { + id: string + label: string + icon?: ReactNode + after?: WorkloadTabType + replaces?: WorkloadTabType + render: () => ReactNode +} + export interface ResourceOwnershipContext { application?: { key: string @@ -286,6 +296,24 @@ interface WorkloadViewProps { initialContainer: string | null onConsumeInitialContainer: () => void }) => ReactNode + /** + * A composed summary for kinds that have one. When it returns content, the + * Overview tab shows the summary and the resource's own renderer moves to a + * "Spec & status" tab, so the same facts never render twice. Return null to + * keep the default Overview. + */ + renderSummary?: (props: { + kind: string + apiKind: string + namespace: string + name: string + group?: string + resource: any + context: 'drawer' | 'expanded' + onNavigate?: NavigateToResource + }) => ReactNode + /** Extra tabs for the expanded view (e.g. a domain's own sections). */ + extraTabs?: WorkloadExtraTab[] /** Render a full replacement for the expanded Overview tab. */ renderExpandedOverview?: (props: { kind: string @@ -434,6 +462,8 @@ export function WorkloadView({ renderDiagnoseTab, reachableVia, renderExpandedOverview, + renderSummary, + extraTabs, renderRelatedYaml, renderMetricsTab, renderCostTab, @@ -480,8 +510,10 @@ export function WorkloadView({ // Collapsed mode state (YAML toggle for drawer mode) const [showYaml, setShowYaml] = useState(initialTab === 'yaml') + const [drawerSpec, setDrawerSpec] = useState(false) useEffect(() => { setShowYaml(initialTab === 'yaml') + setDrawerSpec(false) }, [kindProp, namespace, name, initialTab]) const switchView = useCallback((yaml: boolean) => { @@ -776,8 +808,12 @@ export function WorkloadView({ const podEvidenceLoading = resourceLoading || workloadPodsLoading || eventsLoading const logsFallbackReady = !renderLogsTab || (!logsTabVisible && !podEvidenceLoading) const requestedTab: TabType = activeTab + const summaryContext = { kind, apiKind, namespace, name, group, resource, onNavigate: onNavigateToResource } + const expandedSummary = expanded && resource ? renderSummary?.({ ...summaryContext, context: 'expanded' }) ?? null : null + const drawerSummary = !expanded && resource ? renderSummary?.({ ...summaryContext, context: 'drawer' }) ?? null : null const tabs: DetailShellTab[] = [ { id: 'overview', label: 'Overview', icon: }, + { id: 'spec', label: 'Spec & status', icon: , hidden: !expandedSummary }, { id: 'topology', label: 'Topology', icon: , hidden: topologyTabHidden }, { id: 'timeline', @@ -796,12 +832,15 @@ export function WorkloadView({ { id: 'cost', label: 'Cost', icon: , hidden: !costTabVisible }, { id: 'yaml', label: 'YAML', icon: }, ] - const requestedTabAvailable = tabs.some((tab) => tab.id === requestedTab && !tab.hidden) + const allTabs = mergeExtraTabs(tabs, expanded ? extraTabs : undefined) + const requestedTabAvailable = allTabs.some((tab) => tab.id === requestedTab && !tab.hidden) const effectiveTab: TabType = requestedTabAvailable ? requestedTab : 'overview' + const activeExtraTab = expanded ? extraTabs?.find((x) => x.id === effectiveTab) : undefined const shouldCommitFallback = requestedTab !== 'overview' && !requestedTabAvailable && ( + (requestedTab === 'spec' && !!resource && !resourceLoading && !expandedSummary) || (requestedTab === 'topology' && topologyTabHidden) || (requestedTab === 'metrics' && (!renderMetricsTab || (!!resource && !resourceLoading && !showMetricsTab))) || (requestedTab === 'cost' && (!renderCostTab || (!!resource && !resourceLoading && !showCostTab))) || @@ -941,6 +980,33 @@ export function WorkloadView({ /> ) : ( + {drawerSummary && ( +
+
+ {([['overview', 'Overview'], ['spec', 'Spec & status']] as const).map(([id, label]) => { + const on = id === 'spec' ? drawerSpec : !drawerSpec + return ( + + ) + })} +
+
+ )} + {drawerSummary && !drawerSpec ? ( + drawerSummary + ) : ( + <> {renderOverviewLead && hasOperationalIssues && (
{renderOverviewLead({ kind, namespace, name })} @@ -969,6 +1035,8 @@ export function WorkloadView({ updatesError={resourceFocusedUpdatesError} mainFooter={renderOverviewExtra && renderOverviewExtra({ kind, namespace, name, group, context: 'drawer' })} /> + + )} )}
@@ -1085,7 +1153,7 @@ export function WorkloadView({ )} } - tabs={tabs} + tabs={allTabs} activeTab={effectiveTab} onTabChange={handleSetTab} scopeControls={scopeControls} @@ -1100,7 +1168,10 @@ export function WorkloadView({ )}
- {effectiveTab === 'overview' && expandedOverview ? ( + {activeExtraTab &&
{activeExtraTab.render()}
} + {effectiveTab === 'overview' && expandedSummary ? ( +
{expandedSummary}
+ ) : effectiveTab === 'overview' && expandedOverview ? (
{hasOperationalIssues && renderOverviewLead && (
@@ -1109,7 +1180,7 @@ export function WorkloadView({ )} {expandedOverview}
- ) : effectiveTab === 'overview' && ( + ) : (effectiveTab === 'overview' || effectiveTab === 'spec') && ( [], extra: WorkloadExtraTab[] | undefined): DetailShellTab[] { + if (!extra || extra.length === 0) return tabs + const replaced = new Set(extra.map((x) => x.replaces).filter(Boolean)) + const out = tabs.map((t) => (replaced.has(t.id) ? { ...t, hidden: true } : t)) + for (const x of extra) { + const tab: DetailShellTab = { id: x.id as TabType, label: x.label, icon: x.icon } + const anchor = x.after ?? x.replaces + const idx = anchor ? out.findIndex((t) => t.id === anchor) : -1 + if (idx >= 0) out.splice(idx + 1, 0, tab) + else out.push(tab) + } + return out +} diff --git a/packages/k8s-ui/src/components/workload/index.ts b/packages/k8s-ui/src/components/workload/index.ts index 1d71d59b16..693176b3bb 100644 --- a/packages/k8s-ui/src/components/workload/index.ts +++ b/packages/k8s-ui/src/components/workload/index.ts @@ -8,5 +8,6 @@ export { type ResourceOwnershipContext, type ServingResourceDetail, type WorkloadTabType, + type WorkloadExtraTab, } from './WorkloadView' export { ResourceDetailDrawer } from './ResourceDetailDrawer' diff --git a/packages/k8s-ui/src/components/workload/workload-logs-availability.test.ts b/packages/k8s-ui/src/components/workload/workload-logs-availability.test.ts index c78a3fd605..c34b02a022 100644 --- a/packages/k8s-ui/src/components/workload/workload-logs-availability.test.ts +++ b/packages/k8s-ui/src/components/workload/workload-logs-availability.test.ts @@ -17,4 +17,10 @@ describe('supportsLogsWithoutPods', () => { expect(supportsLogsWithoutPods('jobs', 'Job', 'batch', 'batch/v1')).toBe(true) expect(supportsLogsWithoutPods('jobs', 'Job', 'example.io', 'example.io/v1')).toBe(false) }) + + it('offers a CloudNativePG Cluster its merged instance logs, but not other Cluster kinds', () => { + expect(supportsLogsWithoutPods('clusters', 'Cluster', 'postgresql.cnpg.io', 'postgresql.cnpg.io/v1')).toBe(true) + expect(supportsLogsWithoutPods('clusters', 'Cluster', undefined, 'postgresql.cnpg.io/v1')).toBe(true) + expect(supportsLogsWithoutPods('clusters', 'Cluster', 'cluster.x-k8s.io', 'cluster.x-k8s.io/v1beta1')).toBe(false) + }) }) diff --git a/packages/k8s-ui/src/index.ts b/packages/k8s-ui/src/index.ts index d38efd785e..7354c75027 100644 --- a/packages/k8s-ui/src/index.ts +++ b/packages/k8s-ui/src/index.ts @@ -55,6 +55,9 @@ export * from './components/checks' // queue) export * from './components/issues' +// CloudNativePG workspace model (fleet derivation over /api/cnpg/workspace) +export * from './components/cnpg' + // Cluster switcher (shared trigger+dropdown for OSS Radar and Radar Hub) export * from './components/cluster-switcher' diff --git a/packages/k8s-ui/src/utils/log-level.ts b/packages/k8s-ui/src/utils/log-level.ts index 2037158d66..f064f3b551 100644 --- a/packages/k8s-ui/src/utils/log-level.ts +++ b/packages/k8s-ui/src/utils/log-level.ts @@ -49,6 +49,12 @@ export function normalizeLevel(raw: unknown): LogLevel | null { return 'unknown' } +// PostgreSQL's error_severity values (every DEBUGn is written as DEBUG); +// anything else under `record` isn't PostgreSQL's. +const POSTGRES_SEVERITIES: Record = { + DEBUG: 'debug', LOG: 'info', INFO: 'info', NOTICE: 'info', WARNING: 'warn', ERROR: 'error', FATAL: 'error', PANIC: 'error', +} + const LEVEL_FIELD_KEYS = ['level', 'lvl', 'severity', 'levelname', 'log.level'] as const /** @@ -57,6 +63,14 @@ const LEVEL_FIELD_KEYS = ['level', 'lvl', 'severity', 'levelname', 'log.level'] * come from the same field. */ export function selectLevelField(obj: Record): { raw: unknown; level: LogLevel } | null { + // CloudNativePG wraps each PostgreSQL line in an instance-manager record + // whose own level is usually info; the database's severity is nested. + const pgRecord = obj.record + if (pgRecord && typeof pgRecord === 'object' && !Array.isArray(pgRecord)) { + const raw = (pgRecord as Record).error_severity + const level = typeof raw === 'string' ? POSTGRES_SEVERITIES[raw.trim().toUpperCase()] : undefined + if (level) return { raw, level } + } for (const key of LEVEL_FIELD_KEYS) { const level = normalizeLevel(obj[key]) if (level) return { raw: obj[key], level } diff --git a/pkg/timeline/converter.go b/pkg/timeline/converter.go index c59ae83116..50ca141d95 100644 --- a/pkg/timeline/converter.go +++ b/pkg/timeline/converter.go @@ -9,6 +9,7 @@ import ( corev1 "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" ) // informerEventID derives a deterministic id from the resource's observable @@ -183,9 +184,6 @@ func ExtractLabels(obj any) map[string]string { } allLabels := meta.GetLabels() - if len(allLabels) == 0 { - return nil - } // Only keep labels that are useful for grouping. The GitOps identity // labels must ride along or the app-membership matchKeys the server ships @@ -214,6 +212,9 @@ func ExtractLabels(obj any) map[string]string { relevant[key] = v } } + if cluster := cnpgOwningCluster(obj, allLabels); cluster != "" { + relevant[CNPGClusterLabel] = cluster + } if len(relevant) == 0 { return nil @@ -221,6 +222,35 @@ func ExtractLabels(obj any) map[string]string { return relevant } +// CNPGClusterLabel is the retained label naming the CloudNativePG Cluster a +// row's subject belongs to. +const CNPGClusterLabel = "cnpg.io/cluster" + +// cnpgOwningCluster names the CloudNativePG Cluster an object belongs to, so a +// child's rows stay attributable after it is deleted. CNPG children that +// reference their Cluster through spec.cluster.name (Backup, Pooler, Database, +// ...) are not all labelled; the operator only labels what it creates. +func cnpgOwningCluster(obj any, labels map[string]string) string { + switch o := obj.(type) { + case *corev1.Pod: + return labels[CNPGClusterLabel] + case *unstructured.Unstructured: + group := resourceid.GroupFromAPIVersion(o.GetAPIVersion()) + if group == "" && o.GetKind() == "Pod" { + return labels[CNPGClusterLabel] + } + if group != "postgresql.cnpg.io" && group != "barmancloud.cnpg.io" { + return "" + } + if v := labels[CNPGClusterLabel]; v != "" { + return v + } + name, _, _ := unstructured.NestedString(o.Object, "spec", "cluster", "name") + return name + } + return "" +} + // Resource health classification for timeline events lives with the canonical // classifiers in internal/k8s (classifyTimelineHealth → ClassifyPodHealth), not // here: the timeline package can't reach that logic across the module boundary, diff --git a/pkg/timeline/converter_cnpg_test.go b/pkg/timeline/converter_cnpg_test.go new file mode 100644 index 0000000000..47338170cd --- /dev/null +++ b/pkg/timeline/converter_cnpg_test.go @@ -0,0 +1,66 @@ +package timeline + +import ( + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" +) + +func cnpgTestObject(apiVersion, kind string, labels map[string]string, spec map[string]any) *unstructured.Unstructured { + u := &unstructured.Unstructured{Object: map[string]any{"apiVersion": apiVersion, "kind": kind, "metadata": map[string]any{"name": "x", "namespace": "db"}}} + if spec != nil { + u.Object["spec"] = spec + } + if labels != nil { + u.SetLabels(labels) + } + return u +} + +func TestExtractLabelsRetainsTheOwningCNPGCluster(t *testing.T) { + clusterRef := map[string]any{"cluster": map[string]any{"name": "pg-orders"}} + cases := []struct { + name string + obj any + want string + }{ + {"unlabelled Backup records spec.cluster.name", cnpgTestObject("postgresql.cnpg.io/v1", "Backup", nil, clusterRef), "pg-orders"}, + {"every spec.cluster kind", cnpgTestObject("postgresql.cnpg.io/v1", "Database", map[string]string{"team": "a"}, clusterRef), "pg-orders"}, + {"the label wins over spec", cnpgTestObject("postgresql.cnpg.io/v1", "Pooler", map[string]string{"cnpg.io/cluster": "pg-labelled"}, clusterRef), "pg-labelled"}, + {"barman ObjectStore label", cnpgTestObject("barmancloud.cnpg.io/v1", "ObjectStore", map[string]string{"cnpg.io/cluster": "pg-orders"}, nil), "pg-orders"}, + {"a Velero Backup is not attributed", cnpgTestObject("velero.io/v1", "Backup", map[string]string{"cnpg.io/cluster": "pg-orders"}, clusterRef), ""}, + {"a Deployment keeps no cnpg label", cnpgTestObject("apps/v1", "Deployment", map[string]string{"cnpg.io/cluster": "pg-orders"}, nil), ""}, + {"CNPG object without a cluster", cnpgTestObject("postgresql.cnpg.io/v1", "ImageCatalog", nil, nil), ""}, + {"typed Pod", &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "pg-orders-1", Labels: map[string]string{"cnpg.io/cluster": "pg-orders"}}}, "pg-orders"}, + {"unstructured Pod", cnpgTestObject("v1", "Pod", map[string]string{"cnpg.io/cluster": "pg-orders"}, nil), "pg-orders"}, + {"a Pod's spec is never read", cnpgTestObject("v1", "Pod", nil, clusterRef), ""}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := ExtractLabels(tc.obj)[CNPGClusterLabel] + if got != tc.want { + t.Fatalf("cnpg.io/cluster = %q, want %q (labels %v)", got, tc.want, ExtractLabels(tc.obj)) + } + }) + } +} + +func TestExtractLabelsStaysNilWithoutAnythingToRetain(t *testing.T) { + if got := ExtractLabels(cnpgTestObject("apps/v1", "Deployment", nil, nil)); got != nil { + t.Fatalf("labels = %v, want nil", got) + } + if got := ExtractLabels(cnpgTestObject("apps/v1", "Deployment", map[string]string{"unrelated": "x"}, nil)); got != nil { + t.Fatalf("labels = %v, want nil", got) + } +} + +// A deleted child's tombstone keeps the attribution, so K8s Events that arrive +// after the delete still name the Cluster. +func TestTombstoneEntryCarriesCNPGAttribution(t *testing.T) { + entry, ok := ExtractTombstoneEntry(cnpgTestObject("postgresql.cnpg.io/v1", "Backup", nil, map[string]any{"cluster": map[string]any{"name": "pg-orders"}})) + if !ok || entry.Labels[CNPGClusterLabel] != "pg-orders" { + t.Fatalf("tombstone labels = %v ok=%v", entry.Labels, ok) + } +} diff --git a/web/src/App.tsx b/web/src/App.tsx index a52c3d1d12..db13926d3a 100644 --- a/web/src/App.tsx +++ b/web/src/App.tsx @@ -38,6 +38,9 @@ import { CloudFunnelButton } from './components/CloudFunnelButton' import { useNavCustomization } from './context/NavCustomization' import type { FleetTakeoverTarget } from './context/NavCustomization' import { PrimaryNavRail } from './components/nav/PrimaryNavRail' +import { CNPGView } from './components/cnpg/CNPGView' +import { CNPG_SCREENS, cnpgDetailKindFor, cnpgDetailPath, parseCNPGRoute } from './components/cnpg/routes' +import { currentPageLabel } from './components/cnpg/paths' import { navigateFromPrimaryRail } from './components/nav/navigation' import { useNavRailPinned } from './hooks/useNavRailPinned' import { useMediaQuery } from './hooks/useMediaQuery' @@ -124,7 +127,7 @@ const FLEET_MODE_KINDS = new Set([ // Convert API resource name back to topology node ID prefix // Extended MainView type that includes traffic and cost -type ExtendedMainView = MainView | 'traffic' | 'cost' | 'capacity' | 'workload' | 'checks' | 'gitops' | 'compare' | 'helmCompare' | 'issues' | 'applications' | 'investigations' +type ExtendedMainView = MainView | 'traffic' | 'cost' | 'capacity' | 'cnpg' | 'workload' | 'checks' | 'gitops' | 'compare' | 'helmCompare' | 'issues' | 'applications' | 'investigations' // Extract view from URL path function getViewFromPath(pathname: string): ExtendedMainView { @@ -138,6 +141,7 @@ function getViewFromPath(pathname: string): ExtendedMainView { if (path === 'traffic') return 'traffic' if (path === 'cost') return 'cost' if (path === 'capacity') return 'capacity' + if (path === 'cnpg') return 'cnpg' if (path === 'workload') return 'workload' if (path === 'checks' || path === 'audit') return 'checks' // /audit = legacy → checks if (path === 'gitops') return 'gitops' @@ -165,7 +169,7 @@ function usageView(pathname: string, view: ExtendedMainView, upgrade: boolean): const CRASH_LABELS: Record = { home: 'Home', topology: 'Topology', resources: 'Resources', timeline: 'Timeline', issues: 'Issues', helm: 'Helm', helmCompare: 'HelmCompare', traffic: 'Traffic', - cost: 'Cost', capacity: 'Capacity', checks: 'Checks', gitops: 'GitOps', + cost: 'Cost', capacity: 'Capacity', cnpg: 'CloudNativePG', checks: 'Checks', gitops: 'GitOps', applications: 'Applications', workload: 'Workload', compare: 'Compare', investigations: 'Investigations', } @@ -282,6 +286,12 @@ function radarPageTitle(pathname: string, search = '', apiResources?: APIResourc if (pathSegments[1] === 'activity') return 'Capacity Activity' } + if (view === 'cnpg') { + const route = parseCNPGRoute(pathname) + if (route.detail) return route.detail.name + const screen = CNPG_SCREENS.find((s) => s.id === route.screen) + return `CloudNativePG ${screen?.label ?? 'Overview'}` + } if (view === 'home') return 'Overview' // Every other view's label is its id capitalized — getViewFromPath has already // normalized aliases (e.g. /audit → 'checks'), so no lookup table is needed. @@ -875,7 +885,7 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL gitops: 'g o', checks: 'g u', cost: 'g c', capacity: 'g p', // Non-rail views (reachable via deep links / actions, not the rail) get no // dedicated mnemonic — listed for exhaustiveness so the type stays total. - workload: '', compare: '', helmCompare: '', investigations: '', + workload: '', compare: '', helmCompare: '', investigations: '', cnpg: '', } const views = Object.keys(VIEW_SHORTCUT_KEYS).filter( (v): v is ExtendedMainView => VIEW_SHORTCUT_KEYS[v as ExtendedMainView] !== '', @@ -1064,6 +1074,7 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL secretsChanged: boolean timer: number | null }>({ changedKinds: new Set(), structuralKinds: new Set(), environmentNamespaces: new Set(), environmentPods: new Map(), secretsChanged: false, timer: null }) + const cnpgInvalidationPendingRef = useRef(false) const slowInvalidationRef = useRef<{ updatedKinds: Set // update-only churn → throttled list + dashboard timer: number | null @@ -1095,6 +1106,8 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL const structural = event.operation === 'add' || event.operation === 'delete' const applicationWorkload = ['deployments', 'statefulsets', 'daemonsets', 'rollouts'].includes(kind) + if (event.group?.endsWith('.cnpg.io')) cnpgInvalidationPendingRef.current = true + const fast = fastInvalidationRef.current fast.changedKinds.add(kind) if (structural) fast.structuralKinds.add(kind) @@ -1138,6 +1151,10 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL // GitOps view is mounted (Phase 2 will make this relevance-aware). queryClient.invalidateQueries({ queryKey: ['gitops-tree'] }) queryClient.invalidateQueries({ queryKey: ['gitops-insights'] }) + if (cnpgInvalidationPendingRef.current) { + cnpgInvalidationPendingRef.current = false + queryClient.invalidateQueries({ queryKey: ['cnpg', 'workspace'] }) + } fastInvalidationRef.current = { changedKinds: new Set(), structuralKinds: new Set(), environmentNamespaces: new Set(), environmentPods: new Map(), secretsChanged: false, timer: null } }, 3000) } @@ -1223,6 +1240,10 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL const nextParams = new URLSearchParams() const diagnoseRun = new URLSearchParams(location.search).get('ai-run') if (diagnoseRun) nextParams.set('ai-run', diagnoseRun) + // A CNPG detail keeps the context it belongs to, so it can say it is + // not in the new one instead of loading a same-named object. + const pinnedCtx = new URLSearchParams(location.search).get('ctx') + if (pinnedCtx && location.pathname.startsWith('/cnpg/')) nextParams.set('ctx', pinnedCtx) navigate( { pathname: location.pathname, search: nextParams.toString() }, { replace: true, state: location.state }, @@ -1749,7 +1770,7 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL setVisibleKinds(new Set()) }, []) - const navActiveView = mainView === 'helmCompare' ? 'helm' : mainView + const navActiveView = mainView === 'helmCompare' ? 'helm' : mainView === 'cnpg' ? 'resources' : mainView return ( @@ -2302,6 +2323,16 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL )} + {!viewsSyncGated && mainView === 'cnpg' && ( + setSelectedResource(null)} + onClearNamespaces={clearAllNamespaces} + /> + )} + {/* Takeover splash. When the host claims the current view via fleetTakeoverHref, the redirect effect above is mid-flight — render a brief splash instead of the inline view (which would flash + fire its @@ -2369,7 +2400,15 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL onClose={closeDrawer} onNavigate={(res) => navigateToResource(res)} canCollapseToDrawer={!isMobile} - onExpand={(_res, opts) => { + onExpand={(res, opts) => { + const cnpgPlural = cnpgDetailKindFor(res.kind, res.group) + if (cnpgPlural) { + navigate( + cnpgDetailPath({ plural: cnpgPlural, namespace: res.namespace, name: res.name }, connection.context || undefined, opts?.yaml ? 'yaml' : undefined), + { state: { returnLabel: currentPageLabel(), returnCtx: connection.context } }, + ) + return + } // Grow the peek into a fullscreen overlay (?full=1, pushed so Back // collapses) over whatever view is underneath — list, topology graph, // GitOps, Applications — which stays mounted. Carry the YAML tab when diff --git a/web/src/api/cnpg.ts b/web/src/api/cnpg.ts new file mode 100644 index 0000000000..89a560edfd --- /dev/null +++ b/web/src/api/cnpg.ts @@ -0,0 +1,95 @@ +import { useQuery } from '@tanstack/react-query' +import type { CNPGWorkspaceResponse, TimelineEvent } from '@skyhook-io/k8s-ui' +import { fetchJSON } from './client' + +// /api/cnpg/workspace +// +// Every CloudNativePG object the caller may read, with per-kind coverage, the +// CNPG issues on them and the no-schedule audit finding. One query feeds the +// workspace screens, the Resources sidebar counts and the composed summaries, +// so they can never disagree about what they count. +export function useCNPGWorkspace(namespaces: string[], options?: { enabled?: boolean }) { + const ns = [...namespaces].sort().join(',') + return useQuery({ + queryKey: ['cnpg', 'workspace', ns], + queryFn: ({ signal }) => fetchJSON(`/cnpg/workspace${ns ? `?namespaces=${encodeURIComponent(ns)}` : ''}`, signal), + enabled: options?.enabled ?? true, + staleTime: 10_000, + refetchInterval: 30_000, + placeholderData: (prev) => prev, + }) +} + +export interface CNPGOperatorCoverage { + state: 'full' | 'partial' | 'denied' | 'syncing' | 'error' + deniedNamespaces?: string[] +} + +export interface CNPGOperatorComponent { + role: 'operator' | 'plugin' + pluginName?: string + namespace: string + deployment: string + image?: string + version?: string + readyReplicas: number | null + replicas: number | null +} + +export interface CNPGOperatorConfig { + kind: 'ConfigMap' | 'Secret' + namespace: string + name: string + purpose: 'operator' | 'monitoring' + exists?: boolean | null + readable?: boolean + reason?: string + data?: Record +} + +export interface CNPGOperatorResponse { + coverage: { deployments: CNPGOperatorCoverage; services: CNPGOperatorCoverage } + components: CNPGOperatorComponent[] + config: CNPGOperatorConfig[] +} + +// /api/cnpg/operator +// +// Operator and plugin workloads plus where the operator's configuration lives. +// Deliberately not filtered by the namespace view filter: the operator runs in +// its own namespace, which users rarely have selected. +export function useCNPGOperator(options?: { enabled?: boolean }) { + return useQuery({ + queryKey: ['cnpg', 'operator'], + queryFn: ({ signal }) => fetchJSON('/cnpg/operator', signal), + enabled: options?.enabled ?? true, + staleTime: 30_000, + refetchInterval: 60_000, + }) +} + +export interface CNPGClusterActivityResponse { + events: TimelineEvent[] + oldest: string | null + attributionSince: string | null + truncated: boolean +} + +// /api/cnpg/clusters/{ns}/{name}/activity +// +// The Cluster's history together with its instance Pods and every CNPG object +// attributed to it, including ones since deleted. +export function useCNPGClusterActivity(namespace: string, name: string, sinceHours = 24) { + return useQuery({ + queryKey: ['cnpg', 'activity', namespace, name, sinceHours], + queryFn: ({ signal }) => { + const since = new Date(Date.now() - sinceHours * 3600_000).toISOString() + return fetchJSON( + `/cnpg/clusters/${encodeURIComponent(namespace)}/${encodeURIComponent(name)}/activity?since=${encodeURIComponent(since)}&limit=500`, + signal, + ) + }, + staleTime: 10_000, + refetchInterval: 30_000, + }) +} diff --git a/web/src/components/cnpg/CNPGClusterActivity.tsx b/web/src/components/cnpg/CNPGClusterActivity.tsx new file mode 100644 index 0000000000..0cceeeb2f5 --- /dev/null +++ b/web/src/components/cnpg/CNPGClusterActivity.tsx @@ -0,0 +1,49 @@ +import { useState } from 'react' +import { PaneLoader, TimelineList, formatAge, type NavigateToResource } from '@skyhook-io/k8s-ui' +import { useCNPGClusterActivity } from '../../api/cnpg' +import { Notice } from '../capacity/shared' +import { Segments } from './shared' + +const RANGES = [ + { id: '6', label: '6 h' }, + { id: '24', label: '24 h' }, + { id: '168', label: '7 d' }, +] as const + +/** + * Kubernetes events and changes for the Cluster, its instance Pods and the + * CNPG objects attributed to it — including Backups and declarations deleted + * since. Attribution of child objects relies on a label Radar records at + * ingestion, so history older than that is marked incomplete. + */ +export function CNPGClusterActivity({ namespace, name, onNavigate }: { namespace: string; name: string; onNavigate?: NavigateToResource }) { + const [hours, setHours] = useState('24') + const q = useCNPGClusterActivity(namespace, name, Number(hours)) + + return ( +
+
+ ({ id: r.id, label: r.label }))} /> + + Kubernetes events and changes for the Cluster, its instances, Backups, Poolers and declarations. + +
+
+ {q.data?.attributionSince + ? `The earliest recorded event linking a Backup, Pooler or declaration to this cluster is ${formatAge(q.data.attributionSince)} old. Deleted child objects from before Radar recorded that link are not shown.` + : 'Radar has not recorded any Backup, Pooler or declaration events linked to this cluster yet, so deleted child objects may be missing.'} + {' '}Events for kinds you cannot list are omitted. +
+ {q.data?.truncated && Showing the most recent events only; narrow the range to see all of them.} + {q.error && !q.data ? ( + Activity could not be loaded: {q.error instanceof Error ? q.error.message : 'unknown error'} + ) : !q.data ? ( + + ) : ( +
+ +
+ )} +
+ ) +} diff --git a/web/src/components/cnpg/CNPGClusterLogs.tsx b/web/src/components/cnpg/CNPGClusterLogs.tsx new file mode 100644 index 0000000000..974f6be4ea --- /dev/null +++ b/web/src/components/cnpg/CNPGClusterLogs.tsx @@ -0,0 +1,62 @@ +import { useCallback } from 'react' +import { useSearchParams } from 'react-router-dom' +import { WorkloadLogsViewer, type WorkloadLogsFetchParams, type WorkloadLogsResult } from '@skyhook-io/k8s-ui' +import { fetchJSON } from '../../api/client' +import { getApiBase, getCredentialsMode } from '../../api/config' +import { useDesktopDownload } from '../../hooks/useDesktopDownload' +import { useTheme } from '../../context/ThemeContext' + +function logsPath(namespace: string, name: string) { + return `/cnpg/clusters/${encodeURIComponent(namespace)}/${encodeURIComponent(name)}/logs` +} + +function query(params: WorkloadLogsFetchParams, tailDefault?: number) { + const q = new URLSearchParams() + if (params.container) q.set('container', params.container) + const tail = params.tailLines ?? tailDefault + if (tail) q.set('tailLines', String(tail)) + if (params.sinceSeconds) q.set('sinceSeconds', String(params.sinceSeconds)) + const s = q.toString() + return s ? `?${s}` : '' +} + +/** + * Logs merged from every instance Pod of a CloudNativePG Cluster. `?pod=` + * preselects one instance (the fleet's "Logs" action and "Open instance logs"). + */ +export function CNPGClusterLogs({ namespace, name }: { namespace: string; name: string }) { + const [searchParams] = useSearchParams() + const pod = searchParams.get('pod') + const desktopDownload = useDesktopDownload() + const { theme } = useTheme() + + const fetchAll = useCallback( + (params: WorkloadLogsFetchParams) => + fetchJSON(`${logsPath(namespace, name)}${query(params)}`, { signal: params.signal }), + [namespace, name], + ) + const createStream = useCallback( + (params: WorkloadLogsFetchParams) => + new EventSource(`${getApiBase()}${logsPath(namespace, name)}/stream${query(params, 50)}`, { + withCredentials: getCredentialsMode() === 'include', + }), + [namespace, name], + ) + + return ( +
+ +
+ ) +} diff --git a/web/src/components/cnpg/CNPGDeclarations.tsx b/web/src/components/cnpg/CNPGDeclarations.tsx new file mode 100644 index 0000000000..24092ad3b8 --- /dev/null +++ b/web/src/components/cnpg/CNPGDeclarations.tsx @@ -0,0 +1,311 @@ +import { useMemo, type ReactNode } from 'react' +import { clsx } from 'clsx' +import { AlertTriangle } from 'lucide-react' +import { Badge, cnpgGitOpsSource, isApiGroup, toneTextClass, type CNPGFleetRow } from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { + CNPGWorkspaceHeader, + CoverageNotice, + FilterChips, + ScreenBody, + Segments, + Sub, + clusterResource, + cnpgResource, + coverageEmpty, + worstCoverage, + namespaceChip, + type CNPGScreenProps, +} from './shared' +import { sameResource } from './routes' + +type State = 'applied' | 'failed' | 'pending' + +interface DeclItem { + key: string + kind: 'Database' | 'Publication' | 'Subscription' | 'Managed role' + pgName: string + indent: boolean + state: State + meta?: string + error?: string + source?: string + resource: SelectedResource + isField: boolean +} + +interface DeclGroup { + namespace: string + cluster: string + row: CNPGFleetRow | null + items: DeclItem[] +} + +function stateOf(obj: any): State { + const applied = obj?.status?.applied + if (applied === true) return 'applied' + if (applied === false) return 'failed' + return 'pending' +} + +function gitopsSource(obj: any): string | undefined { + const src = cnpgGitOpsSource(obj) + if (!src) return undefined + return `${src.tool === 'argocd' ? 'Argo CD' : 'Flux'} ${src.name}` +} + +const STATE_BADGE: Record = { + applied: { severity: 'success', text: 'Applied' }, + failed: { severity: 'warning', text: 'Not applied' }, + pending: { severity: 'neutral', text: 'Pending' }, +} + +function roleState(cluster: any, role: string): { state: State; error?: string } { + const status = cluster?.status?.managedRolesStatus + const errs = status?.cannotReconcile?.[role] + if (Array.isArray(errs) && errs.length > 0) return { state: 'failed', error: errs.join('; ') } + const by = status?.byStatus ?? {} + if ((by.reconciled ?? []).includes(role)) return { state: 'applied' } + return { state: 'pending' } +} + +export function CNPGDeclarations({ data, fleet, namespaces, searchParams, onSetParams, onInspect, inspected, onClearNamespaces }: CNPGScreenProps) { + const clusterFilter = searchParams.get('cluster') + const show = (searchParams.get('show') as 'failed' | 'pending' | null) ?? null + + const groups = useMemo(() => { + const byCluster = new Map() + const group = (ns: string, cluster: string) => { + const k = `${ns}/${cluster}` + let g = byCluster.get(k) + if (!g) { + g = { namespace: ns, cluster, row: fleet.rows.find((r) => r.namespace === ns && r.name === cluster) ?? null, items: [] } + byCluster.set(k, g) + } + return g + } + const valid = (o: any) => isApiGroup(o?.apiVersion, 'postgresql.cnpg.io') + const dbs = (data.objects.databases ?? []).filter(valid) + const pubs = (data.objects.publications ?? []).filter(valid) + const subs = (data.objects.subscriptions ?? []).filter(valid) + const placed = new Set() + for (const d of dbs) { + const ns = d.metadata?.namespace ?? '' + const cluster = d.spec?.cluster?.name ?? '(no cluster)' + const g = group(ns, cluster) + const st = stateOf(d) + g.items.push({ + key: `db/${ns}/${d.metadata?.name}`, + kind: 'Database', + pgName: d.spec?.name ?? d.metadata?.name, + indent: false, + state: st, + meta: d.spec?.owner ? `owner ${d.spec.owner}` : undefined, + error: st === 'failed' ? d.status?.message : undefined, + source: gitopsSource(d), + resource: cnpgResource('databases', ns, d.metadata?.name), + isField: false, + }) + for (const [list, kind] of [[pubs, 'Publication'], [subs, 'Subscription']] as const) { + for (const p of list) { + if (p.metadata?.namespace !== ns || p.spec?.cluster?.name !== d.spec?.cluster?.name || p.spec?.dbname !== d.spec?.name) continue + placed.add(p) + const pst = stateOf(p) + g.items.push({ + key: `${kind}/${ns}/${p.metadata?.name}`, + kind, + pgName: p.spec?.name ?? p.metadata?.name, + indent: true, + state: pst, + meta: + kind === 'Publication' + ? p.spec?.target?.allTables ? 'all tables' : 'selected objects' + : `from ${p.spec?.publicationName ?? '?'} on ${p.spec?.externalClusterName ?? '?'}`, + error: pst === 'failed' ? p.status?.message : undefined, + source: gitopsSource(p), + resource: cnpgResource(kind === 'Publication' ? 'publications' : 'subscriptions', ns, p.metadata?.name), + isField: false, + }) + } + } + } + for (const [list, kind] of [[pubs, 'Publication'], [subs, 'Subscription']] as const) { + for (const p of list) { + if (placed.has(p)) continue + const ns = p.metadata?.namespace ?? '' + const g = group(ns, p.spec?.cluster?.name ?? '(no cluster)') + const pst = stateOf(p) + g.items.push({ + key: `${kind}/${ns}/${p.metadata?.name}`, + kind, + pgName: p.spec?.name ?? p.metadata?.name, + indent: false, + state: pst, + meta: p.spec?.dbname ? `database ${p.spec.dbname}` : undefined, + error: pst === 'failed' ? p.status?.message : undefined, + source: gitopsSource(p), + resource: cnpgResource(kind === 'Publication' ? 'publications' : 'subscriptions', ns, p.metadata?.name), + isField: false, + }) + } + } + for (const row of fleet.rows) { + const roles: any[] = Array.isArray(row.cluster?.spec?.managed?.roles) ? row.cluster.spec.managed.roles : [] + if (roles.length === 0) continue + const g = group(row.namespace, row.name) + for (const r of roles) { + if (!r?.name) continue + const rs = roleState(row.cluster, r.name) + g.items.push({ + key: `role/${row.namespace}/${row.name}/${r.name}`, + kind: 'Managed role', + pgName: r.name, + indent: false, + state: rs.state, + meta: r.ensure === 'absent' ? 'ensure absent' : 'spec.managed.roles', + error: rs.error, + source: `Cluster ${row.name} spec`, + resource: clusterResource(row.namespace, row.name), + isField: true, + }) + } + } + return [...byCluster.values()] + .filter((g) => !clusterFilter || `${g.namespace}/${g.cluster}` === clusterFilter) + .map((g) => ({ ...g, items: show ? g.items.filter((i) => i.state === show) : g.items })) + .filter((g) => g.items.length > 0) + .sort((a, b) => { + const fa = a.items.some((i) => i.state === 'failed') ? 0 : 1 + const fb = b.items.some((i) => i.state === 'failed') ? 0 : 1 + return fa - fb || a.namespace.localeCompare(b.namespace) || a.cluster.localeCompare(b.cluster) + }) + }, [data.objects.databases, data.objects.publications, data.objects.subscriptions, fleet.rows, clusterFilter, show]) + + const totals = useMemo(() => { + let failed = 0 + let pending = 0 + for (const r of fleet.rows) { + failed += r.declarations.failed + pending += r.declarations.pending + } + return { failed, pending } + }, [fleet.rows]) + const declCoverage = worstCoverage(data.coverage.databases, data.coverage.publications, data.coverage.subscriptions) + + const chips = [ + ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), + ...namespaceChip(namespaces, onClearNamespaces), + ] + + return ( +
+ + + +
+ onSetParams({ show: id === 'all' ? null : id })} + options={[ + { id: 'all', label: 'All declarations' }, + { id: 'failed', label: 'Not applied', count: totals.failed }, + { id: 'pending', label: 'Pending', count: totals.pending }, + ]} + /> +
+ + + {groups.length === 0 ? ( +
+ {show === 'failed' + ? 'No declaration in this scope is reported as not applied.' + : show === 'pending' + ? 'No declaration in this scope is waiting for the operator.' + : coverageEmpty(declCoverage, 'declarations')} + {show && declCoverage?.state !== 'full' ? ' Some declarations are not readable with your access.' : ''} +
+ ) : ( + groups.map((g) => { + const failed = g.items.filter((i) => i.state === 'failed').length + return ( +
+
+ {g.row ? ( + + ) : ( + + {g.cluster} + + )} + {g.namespace} + 0 ? toneTextClass('degraded') : 'text-theme-text-tertiary')}> + {g.items.length} {g.items.length === 1 ? 'declaration' : 'declarations'} + {failed > 0 ? ` · ${failed} not reconciled` : ''} + {!g.row ? ' · target cluster not visible' : ''} + +
+
+ {g.items.map((i) => ( + onInspect(i.resource)} + /> + ))} +
+
+ ) + }) + )} +
+
+ ) +} + +function DeclarationRow({ item, active, onInspect }: { item: DeclItem; active: boolean; onInspect: () => void }) { + const badge = STATE_BADGE[item.state] + let detail: ReactNode = null + if (item.error) { + detail = ( +
+ + Controller error: {item.error} +
+ ) + } + return ( +
{ if (e.key === 'Enter') onInspect() }} + className={clsx( + 'grid cursor-pointer grid-cols-[minmax(0,1.6fr)_minmax(0,0.8fr)_minmax(0,1fr)] gap-x-4 px-4 py-2.5 text-sm transition-colors hover:bg-theme-hover/50', + active && 'selection', + )} + > +
+
+ {item.kind} + {item.pgName} + {item.meta && {item.meta}} +
+ {detail} +
+
+ {badge.text} + {item.isField && field of the Cluster} +
+
+ {item.source ?? GitOps source not recorded} +
+
+ ) +} diff --git a/web/src/components/cnpg/CNPGDetailPage.tsx b/web/src/components/cnpg/CNPGDetailPage.tsx new file mode 100644 index 0000000000..9d56bfd521 --- /dev/null +++ b/web/src/components/cnpg/CNPGDetailPage.tsx @@ -0,0 +1,228 @@ +import { useCallback, useEffect, useMemo } from 'react' +import { useLocation, useNavigate, useSearchParams } from 'react-router-dom' +import { Activity, ArrowLeft, Database, ShieldCheck, Unplug } from 'lucide-react' +import type { WorkloadExtraTab } from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { useConnection } from '../../context/ConnectionContext' +import { useContexts } from '../../api/client' +import { useContextSwitchFlow } from '../useContextSwitchFlow' +import { WorkloadView } from '../workload/WorkloadView' +import { EmptyState } from '../capacity/shared' +import { CNPGClusterActivity } from './CNPGClusterActivity' +import { CNPGProtection } from './CNPGProtection' +import { CNPGScreenGate } from './shared' +import { CNPG_DETAIL_KINDS, CNPG_SCREENS, cnpgDetailKindFor, cnpgDetailPath, cnpgScreenPath, type CNPGDetailTarget } from './routes' +import { currentPageLabel } from './paths' +import { useCNPGFleet } from './useCNPGSidebarWorkspace' + +interface ReturnState { + returnLabel?: string + returnCtx?: string +} + +function ClusterProtectionTab({ namespace, name, onInspect }: { namespace: string; name: string; onInspect: (r: SelectedResource) => void }) { + const { query, fleet } = useCNPGFleet([namespace]) + const [searchParams] = useSearchParams() + return ( + + {(data, readyFleet) => ( + {}} + onInspect={onInspect} + inspected={null} + onClearNamespaces={() => {}} + scopeCluster={{ namespace, name }} + /> + )} + + ) +} + +/** + * The full detail of a CloudNativePG object, framed by the workspace: the + * Resources sidebar keeps the workspace destination highlighted, a return + * control goes back to the task the user came from, and the crumb names the + * object's place. The object's own sections come from Radar's detail view. + * + * `ctx` in the URL pins the Kubernetes context; when the active context is a + * different one, the page says so instead of loading a same-named object. + */ +export function CNPGDetailPage({ + target, + namespaces, + onOpenResource, +}: { + target: CNPGDetailTarget + namespaces: string[] + onOpenResource: (resource: SelectedResource) => void +}) { + const location = useLocation() + const navigate = useNavigate() + const [searchParams, setSearchParams] = useSearchParams() + const { connection } = useConnection() + const activeContext = connection.context + const pinnedContext = searchParams.get('ctx') + + // A link opened without a context belongs to the one active now; pin it so a + // later context switch shows "not in this context" instead of a same-named + // object from the other cluster. + useEffect(() => { + if (pinnedContext || !activeContext) return + const params = new URLSearchParams(searchParams) + params.set('ctx', activeContext) + setSearchParams(params, { replace: true, state: location.state }) + }, [pinnedContext, activeContext]) // eslint-disable-line react-hooks/exhaustive-deps + const returnState = (location.state ?? {}) as ReturnState + const spec = CNPG_DETAIL_KINDS[target.plural] + const home = CNPG_SCREENS.find((s) => s.id === spec.home)! + const returnLabel = returnState.returnLabel && (!returnState.returnCtx || returnState.returnCtx === activeContext) ? returnState.returnLabel : null + + const openRelated = useCallback( + (res: SelectedResource) => { + const plural = cnpgDetailKindFor(res.kind, res.group) + if (plural) { + navigate(cnpgDetailPath({ plural, namespace: res.namespace, name: res.name }, activeContext), { + state: { returnLabel: currentPageLabel(), returnCtx: activeContext } satisfies ReturnState, + }) + } else { + onOpenResource(res) + } + }, + [navigate, activeContext, onOpenResource], + ) + + const extraTabs = useMemo(() => { + if (target.plural !== 'clusters') return undefined + return [ + { + id: 'protection', + label: 'Protection', + icon: , + after: 'spec', + render: () => , + }, + { + id: 'activity', + label: 'Activity', + icon: , + replaces: 'timeline', + render: () => ( + + ), + }, + ] + }, [target.plural, target.namespace, target.name, onOpenResource, openRelated]) + + if (pinnedContext && activeContext && pinnedContext !== activeContext) { + return + } + + const outsideFilter = namespaces.length > 0 && !!target.namespace && !namespaces.includes(target.namespace) + + const breadcrumb = ( +
+ {returnLabel && ( + <> + + + + )} + + {outsideFilter && ( + + Namespace {target.namespace} is outside your namespace filter; this object stays open. + + )} +
+ ) + + return ( +
+ (returnLabel ? navigate(-1) : navigate(home.path))} + hideBackButton + breadcrumb={breadcrumb} + onNavigateToResource={openRelated} + extraTabs={extraTabs} + /> +
+ ) +} + +function NotInContext({ + target, + pinnedContext, + activeContext, + homeLabel, + homePath, +}: { + target: CNPGDetailTarget + pinnedContext: string + activeContext: string + homeLabel: string + homePath: string +}) { + const navigate = useNavigate() + const { data: contexts } = useContexts() + const { requestSwitch, confirmDialog } = useContextSwitchFlow() + const pinned = contexts?.find((c) => c.name === pinnedContext) + return ( + <> + + {pinned && ( + + )} + +
+ } + /> + {confirmDialog} + + ) +} diff --git a/web/src/components/cnpg/CNPGDrawerTrail.tsx b/web/src/components/cnpg/CNPGDrawerTrail.tsx new file mode 100644 index 0000000000..55d3d9dfe7 --- /dev/null +++ b/web/src/components/cnpg/CNPGDrawerTrail.tsx @@ -0,0 +1,37 @@ +import { useLocation, useSearchParams } from 'react-router-dom' +import { ArrowLeft } from 'lucide-react' +import { CNPG_KIND_BY_KEY } from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { decodeDrawerTrail, encodeDrawerTrail, sameResource } from './routes' + +const KIND_BY_PLURAL: Record = Object.fromEntries( + Object.values(CNPG_KIND_BY_KEY).map((k) => [k.plural, k.kind]), +) + +/** + * On a CloudNativePG workspace screen the drawer URL carries the chain of + * objects opened from inside the drawer. This is the step back to the previous + * one, shown for whatever kind is open (a Pod or Secret reached from a CNPG + * object included). + */ +export function CNPGDrawerTrailBack({ resource }: { resource: SelectedResource }) { + const location = useLocation() + const [searchParams, setSearchParams] = useSearchParams() + if (!location.pathname.startsWith('/cnpg')) return null + const trail = decodeDrawerTrail(searchParams.get('drawer')) + if (trail.length < 2 || !sameResource(trail[trail.length - 1], resource)) return null + const prev = trail[trail.length - 2] + const back = () => { + const params = new URLSearchParams(searchParams) + params.set('drawer', encodeDrawerTrail(trail.slice(0, -1))) + setSearchParams(params, { replace: true, state: location.state }) + } + return ( +
+ +
+ ) +} diff --git a/web/src/components/cnpg/CNPGOperator.tsx b/web/src/components/cnpg/CNPGOperator.tsx new file mode 100644 index 0000000000..eb1b285a25 --- /dev/null +++ b/web/src/components/cnpg/CNPGOperator.tsx @@ -0,0 +1,234 @@ +import { useMemo } from 'react' +import { Badge, getCNPGImageCatalogEntries, isApiGroup, PaneLoader, Tooltip } from '@skyhook-io/k8s-ui' +import { useCNPGOperator, type CNPGOperatorComponent, type CNPGOperatorConfig } from '../../api/cnpg' +import { Notice } from '../capacity/shared' +import { + CNPGWorkspaceHeader, + CoverageNotice, + coverageEmpty, + worstCoverage, + Mono, + ScreenBody, + SectionTable, + Sub, + cnpgResource, + type CNPGScreenProps, +} from './shared' + +interface CatalogRow { + key: string + kind: 'ImageCatalog' | 'ClusterImageCatalog' + namespace: string + name: string + images: { major: number; image: string }[] + users: { namespace: string; name: string; major?: number }[] +} + +function readiness(c: CNPGOperatorComponent) { + if (c.readyReplicas === null || c.replicas === null) return Unknown + if (c.replicas === 0) return Scaled to 0 + const ok = c.readyReplicas >= c.replicas + return {c.readyReplicas}/{c.replicas} ready +} + +export function CNPGOperator({ data, fleet, onInspect, inspected }: CNPGScreenProps) { + const operator = useCNPGOperator() + + const catalogs = useMemo(() => { + const out: CatalogRow[] = [] + const clusters = fleet.rows + for (const [key, kind] of [['imageCatalogs', 'ImageCatalog'], ['clusterImageCatalogs', 'ClusterImageCatalog']] as const) { + for (const cat of data.objects[key] ?? []) { + if (!isApiGroup(cat.apiVersion, 'postgresql.cnpg.io')) continue + const ns = cat.metadata?.namespace ?? '' + const name = cat.metadata?.name ?? '' + const users = clusters + .filter((r) => { + const ref = r.cluster?.spec?.imageCatalogRef + if (!ref || ref.name !== name) return false + const refKind = ref.kind || 'ImageCatalog' + if (refKind !== kind) return false + return kind === 'ClusterImageCatalog' || r.namespace === ns + }) + .map((r) => ({ namespace: r.namespace, name: r.name, major: r.cluster?.spec?.imageCatalogRef?.major })) + out.push({ key: `${kind}/${ns}/${name}`, kind, namespace: ns, name, images: getCNPGImageCatalogEntries(cat), users }) + } + } + return out + }, [data.objects, fleet.rows]) + + const direct = fleet.rows.filter((r) => !r.cluster?.spec?.imageCatalogRef) + const op = operator.data + const coverageGaps = op + ? (['deployments', 'services'] as const).filter((k) => op.coverage[k]?.state !== 'full') + : [] + + return ( +
+ + + + {operator.isLoading && !op ? ( + + ) : !op ? ( + Operator details could not be loaded{operator.error instanceof Error ? `: ${operator.error.message}` : '.'} + ) : ( + <> + {coverageGaps.length > 0 && ( + + Some workloads are not readable ({coverageGaps.map((k) => `${k}: ${op.coverage[k].state}`).join(', ')}), so an operator or plugin running in those namespaces may be missing below. + + )} + ( + <> +
{c.role === 'operator' ? 'CloudNativePG operator' : c.pluginName ?? 'Plugin'}
+ {c.role === 'operator' ? 'controller manager' : 'CNPG-I plugin'} + + ), + }, + { + header: 'Version', + width: '14%', + cell: (c) => ( + + {c.version ? {c.version} : Unknown} + + ), + }, + { header: 'Ready', width: '14%', cell: readiness }, + { + header: 'Workload', + width: '42%', + cell: (c) => + c.deployment ? ( + <>Deployment {c.deployment}{c.namespace} + ) : ( + <> + No Deployment matches its Service + {c.namespace} + + ), + }, + ]} + rows={op.components} + rowKey={(c) => `${c.namespace}/${c.deployment}/${c.pluginName ?? c.role}`} + rowResource={(c) => (c.deployment ? { kind: 'deployments', group: 'apps', namespace: c.namespace, name: c.deployment } : null)} + onInspect={onInspect} + inspected={inspected} + empty="No operator or plugin Deployments found in the namespaces you can read." + /> + + )} + + <>{c.name}{c.kind} }, + { header: 'Scope', width: '14%', cell: (c) => (c.kind === 'ClusterImageCatalog' ? 'Cluster-wide' : `Namespace ${c.namespace}`) }, + { + header: 'Images', + width: '40%', + cell: (c) => + c.images.length === 0 ? ( + None + ) : ( +
+ {c.images.map((i) => ( +
+ {i.major} + {i.image} +
+ ))} +
+ ), + }, + { + header: 'Used by', + width: '24%', + cell: (c) => + c.users.length === 0 ? ( + No visible cluster + ) : ( + c.users.map((u) => `${u.name}${u.major !== undefined ? ` (${u.major})` : ''}`).join(', ') + ), + }, + ]} + rows={catalogs} + rowKey={(c) => c.key} + rowResource={(c) => cnpgResource(c.kind === 'ImageCatalog' ? 'imagecatalogs' : 'clusterimagecatalogs', c.namespace, c.name)} + onInspect={onInspect} + inspected={inspected} + empty={coverageEmpty(worstCoverage(data.coverage.imageCatalogs, data.coverage.clusterImageCatalogs), 'image catalogs')} + footer={ + <> + Used-by lists only clusters you can see; the catalog detail asks the server for every user. + {direct.length > 0 && <> Not using a catalog (direct imageName): {direct.map((r) => r.name).join(', ')}.} + + } + /> + + {op && op.config.length > 0 && ( +
+

Operator configuration

+
+ {op.config.map((c) => ( + + ))} +
+
+ )} +
+
+ ) +} + +function ConfigBlock({ config, onInspect }: { config: CNPGOperatorConfig; onInspect: CNPGScreenProps['onInspect'] }) { + const open = () => onInspect({ kind: config.kind === 'ConfigMap' ? 'configmaps' : 'secrets', group: '', namespace: config.namespace, name: config.name }) + const title = ( +
+ {config.kind} + + + {config.namespace} · {config.purpose === 'monitoring' ? 'monitoring queries' : 'operator settings'} + +
+ ) + let body + if (config.kind === 'Secret') { + body =
Referenced by the operator. Secret contents are not shown here.
+ } else if (config.exists === false) { + body =
Referenced but does not exist; the operator runs with its defaults.
+ } else if (!config.readable) { + body =
{config.reason ?? 'Not readable with your access.'}
+ } else { + const entries = Object.entries(config.data ?? {}) + body = + entries.length === 0 ? ( +
No keys set.
+ ) : ( +
+ {entries.map(([k, v]) => ( +
+
{k}
+
{v.length > 400 ? `${v.slice(0, 400)}…` : v}
+
+ ))} +
+ ) + } + return ( +
+ {title} + {body} +
+ ) +} diff --git a/web/src/components/cnpg/CNPGOverview.tsx b/web/src/components/cnpg/CNPGOverview.tsx new file mode 100644 index 0000000000..6315aaded3 --- /dev/null +++ b/web/src/components/cnpg/CNPGOverview.tsx @@ -0,0 +1,307 @@ +import { useMemo } from 'react' +import { useNavigate } from 'react-router-dom' +import { clsx } from 'clsx' +import { ArrowRight, Database, FileText, Search } from 'lucide-react' +import { + CNPG_PROBLEM_CATEGORIES, + FactValue, + StatusDot, + Tooltip, + toneTextClass, + type CNPGFleetRow, + type CNPGProblemCategory, +} from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { useConnection } from '../../context/ConnectionContext' +import { EmptyState, ROW_HOVER, TABLE_HEAD, TABLE_WRAP, TBODY, TD, TH } from '../capacity/shared' +import { CNPGWorkspaceHeader, CoverageNotice, FilterChips, type CNPGScreenProps } from './shared' +import { cnpgClusterFullPath, currentPageLabel } from './paths' +import { sameResource } from './routes' + +type Filter = 'attention' | 'all' + +function InstancePills({ row }: { row: CNPGFleetRow }) { + if (row.pods.length === 0) return null + return ( +
+ {row.pods.map((p) => { + const tone = p.ready === true ? 'healthy' : p.ready === false ? 'unhealthy' : 'unknown' + return ( + + + + {p.role === 'primary' ? 'P' : p.role === 'replica' ? 'R' : '?'} + + + ) + })} +
+ ) +} + +function AttentionCell({ row }: { row: CNPGFleetRow }) { + const top = row.problems.find((p) => p.severity !== 'posture') ?? row.problems[0] + if (!top) return — + const tone = top.severity === 'critical' ? 'unhealthy' : top.severity === 'warning' ? 'degraded' : 'neutral' + const more = row.problems.length - 1 + return ( +
+ +
{top.title}
+
+ {more > 0 &&
+{more} more
} +
+ ) +} + +export function CNPGOverview({ + data, + fleet, + namespaces, + searchParams, + onSetParams, + onInspect, + inspected, + onClearNamespaces, +}: CNPGScreenProps) { + const navigate = useNavigate() + const { connection } = useConnection() + const q = searchParams.get('q') ?? '' + const cat = (searchParams.get('cat') as CNPGProblemCategory | null) ?? null + const rawFilter = searchParams.get('filter') as Filter | null + const filter: Filter = rawFilter ?? (fleet.attentionCount > 0 ? 'attention' : 'all') + + const rows = useMemo(() => { + let list = fleet.rows + if (filter === 'attention') list = list.filter((r) => r.attention) + if (cat) list = list.filter((r) => r.categories.has(cat)) + if (q) { + const needle = q.toLowerCase() + list = list.filter((r) => r.name.toLowerCase().includes(needle) || r.namespace.toLowerCase().includes(needle)) + } + return list + }, [fleet, filter, cat, q]) + + const clustersCov = data.coverage.clusters + const total = fleet.rows.length + const context = connection.context || data.context + + if (total === 0) { + const state = clustersCov?.state ?? 'notInstalled' + const empty = + state === 'denied' + ? { title: 'No access to PostgreSQL clusters', detail: 'Your identity cannot list CloudNativePG Clusters. Other CloudNativePG kinds may still be browsable under Resource kinds.' } + : state === 'syncing' + ? { title: 'Loading PostgreSQL clusters', detail: 'Radar is still syncing CloudNativePG Clusters from the API server.' } + : state === 'error' + ? { title: 'PostgreSQL clusters could not be read', detail: 'Reading CloudNativePG Clusters failed; see the Radar server log.' } + : state === 'partial' + ? { title: 'No visible PostgreSQL clusters', detail: 'None in the namespaces you can read. Clusters in namespaces you cannot list are not shown.' } + : namespaces.length > 0 + ? { title: `No PostgreSQL clusters in ${context}`, detail: `None in namespace ${namespaces.join(', ')}. Clear the namespace filter to see the whole cluster.` } + : { title: `No PostgreSQL clusters in ${context}`, detail: 'The CloudNativePG CRDs are installed. Clusters, backups and declarations appear here once they exist.' } + return ( +
+ + 0 ? ( + + ) : undefined + } + /> +
+ ) + } + + const chips: { label: string; onClear: () => void }[] = [] + if (cat) chips.push({ label: `Problem: ${CNPG_PROBLEM_CATEGORIES.find((c) => c.id === cat)?.label ?? cat}`, onClear: () => onSetParams({ cat: null }) }) + if (q) chips.push({ label: `Search: ${q}`, onClear: () => onSetParams({ q: null }) }) + if (namespaces.length > 0) chips.push({ label: `Namespace: ${namespaces.join(', ')}`, onClear: onClearNamespaces }) + + const segment = (id: Filter, label: string, n: number) => { + const on = filter === id + return ( + + ) + } + + const lowerBound = fleet.incompleteKinds.length > 0 + + return ( +
+ + {total} PostgreSQL {total === 1 ? 'cluster' : 'clusters'} · {fleet.attentionCount} + {lowerBound ? '+' : ''} need attention + · {context} + + } + /> +
+
+ + +
+
+ {segment('attention', 'Needs attention', fleet.attentionCount)} + {segment('all', 'All clusters', total)} +
+ {CNPG_PROBLEM_CATEGORIES.filter((c) => fleet.categoryCounts[c.id] > 0).map((c) => { + const on = cat === c.id + return ( + + ) + })} +
+ + onSetParams({ q: e.target.value })} + placeholder="Filter clusters…" + aria-label="Filter clusters" + className="min-w-0 flex-1 bg-transparent text-sm text-theme-text-primary placeholder-theme-text-disabled focus:outline-none" + /> +
+
+ + + +
+
+ + + + + + + + + + + + + + + + + + + + + + + + + {rows.map((row) => { + const ref: SelectedResource = { kind: 'clusters', group: 'postgresql.cnpg.io', namespace: row.namespace, name: row.name } + const active = sameResource(inspected, ref) + return ( + onInspect(ref)} + className={clsx('cursor-pointer', ROW_HOVER, active && 'selection')} + aria-selected={active} + > + + + + + + + + + + ) + })} + +
ClusterReadyReplicationProtectionDeclarationsPGNeeds attentionActions
+
+ p.severity === 'critical') ? 'unhealthy' : 'degraded') : row.controllerStatus.level} /> + {row.name} +
+
{row.namespace}
+
+
+ {row.instances.ready ?? '–'}/{row.instances.desired ?? '–'} + {row.pgVersion ?? '—'} +
+ + +
+
+
+ {rows.length === 0 && ( +
+ {filter === 'attention' && !cat && !q + ? 'No clusters need attention.' + : 'No clusters match these filters.'}{' '} + +
+ )} +
+
+
+
+ ) +} diff --git a/web/src/components/cnpg/CNPGPooling.tsx b/web/src/components/cnpg/CNPGPooling.tsx new file mode 100644 index 0000000000..21c42b50a4 --- /dev/null +++ b/web/src/components/cnpg/CNPGPooling.tsx @@ -0,0 +1,116 @@ +import { useMemo } from 'react' +import { + Badge, + getCNPGPoolerMode, + getCNPGPoolerStatus, + getCNPGPoolerType, + isApiGroup, + type HealthLevel, +} from '@skyhook-io/k8s-ui' +import { + CNPGWorkspaceHeader, + CoverageNotice, + FilterChips, + Mono, + ScreenBody, + SectionTable, + Sub, + cnpgResource, + coverageEmpty, + namespaceChip, + type CNPGScreenProps, +} from './shared' + +const SEVERITY: Record = { + healthy: 'success', + degraded: 'warning', + alert: 'alert', + unhealthy: 'error', + unknown: 'neutral', + neutral: 'neutral', +} + +export function CNPGPooling({ data, fleet, namespaces, searchParams, onSetParams, onInspect, inspected, onClearNamespaces }: CNPGScreenProps) { + const clusterFilter = searchParams.get('cluster') + const poolers = useMemo( + () => + (data.objects.poolers ?? []) + .filter((p) => isApiGroup(p.apiVersion, 'postgresql.cnpg.io')) + .filter((p) => !clusterFilter || `${p.metadata?.namespace}/${p.spec?.cluster?.name}` === clusterFilter), + [data.objects.poolers, clusterFilter], + ) + const chips = [ + ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), + ...namespaceChip(namespaces, onClearNamespaces), + ] + + return ( +
+ + + + + <>{p.metadata?.name}{p.metadata?.namespace} }, + { + header: 'Target cluster', + width: '18%', + cell: (p) => { + const name = p.spec?.cluster?.name + const visible = fleet.rows.some((r) => r.namespace === p.metadata?.namespace && r.name === name) + return ( + <> + {name ?? '—'} + {name && !visible && not visible in this scope} + + ) + }, + }, + { header: 'Type', width: '8%', cell: (p) => {getCNPGPoolerType(p)} }, + { header: 'Mode', width: '12%', cell: (p) => getCNPGPoolerMode(p) }, + { + header: 'Instances', + width: '10%', + cell: (p) => ( + + {typeof p.status?.instances === 'number' ? p.status.instances : '–'}/{typeof p.spec?.instances === 'number' ? p.spec.instances : '–'} + + ), + }, + { + header: 'Status', + width: '14%', + cell: (p) => { + const st = getCNPGPoolerStatus(p) + return {st.text} + }, + }, + { + header: 'Connection pressure', + width: '16%', + cell: () => ( + <> + Not measured + Needs PgBouncer metrics + + ), + }, + ]} + rows={poolers} + rowKey={(p) => `${p.metadata?.namespace}/${p.metadata?.name}`} + rowResource={(p) => cnpgResource('poolers', p.metadata?.namespace, p.metadata?.name)} + onInspect={onInspect} + inspected={inspected} + minWidth={880} + empty={coverageEmpty(data.coverage.poolers, 'Poolers')} + footer="Instances are the Pooler’s own ready count. Client waits and server-pool saturation come from PgBouncer metrics, which Radar does not read yet." + /> + +
+ ) +} diff --git a/web/src/components/cnpg/CNPGProtection.tsx b/web/src/components/cnpg/CNPGProtection.tsx new file mode 100644 index 0000000000..7bbbc0c9e6 --- /dev/null +++ b/web/src/components/cnpg/CNPGProtection.tsx @@ -0,0 +1,320 @@ +import { useMemo } from 'react' +import { + Badge, + FactValue, + formatAge, + formatDuration, + getCNPGBackupStatus, + getCNPGClusterBarmanPlugin, + getCNPGObjectStoreDestination, + getCNPGScheduledBackupStatus, + inferredObjectStoreHealth, + isApiGroup, + toneTextClass, + usersOfObjectStore, + type CNPGFleetRow, + type HealthLevel, +} from '@skyhook-io/k8s-ui' +import { + CNPGWorkspaceHeader, + CoverageNotice, + FilterChips, + Mono, + ScreenBody, + SectionTable, + Sub, + clusterResource, + coverageEmpty, + cnpgResource, + namespaceChip, + type CNPGScreenProps, +} from './shared' + +const SEVERITY: Record = { + healthy: 'success', + degraded: 'warning', + alert: 'alert', + unhealthy: 'error', + unknown: 'neutral', + neutral: 'neutral', +} + +const WEEK_MS = 7 * 24 * 60 * 60 * 1000 + +function backupTime(b: any): string | undefined { + return b?.status?.stoppedAt || b?.status?.startedAt || b?.metadata?.creationTimestamp +} + +function backupStart(b: any): string | undefined { + return b?.status?.startedAt || b?.metadata?.creationTimestamp +} + +function ageText(ts?: string): string { + return ts ? `${formatAge(ts)} ago` : '—' +} + +interface StoreRow { + key: string + namespace: string + name: string + destination: string + users: CNPGFleetRow[] + health: { text: string; tone: HealthLevel; evidence: string } +} + +// The same inference the ObjectStore summary shows, so the two never disagree: +// only the recovery windows of clusters that use the store now count. +function storeHealth(store: any, users: CNPGFleetRow[]): StoreRow['health'] { + const { summary, evidence } = inferredObjectStoreHealth(store, usersOfObjectStore(store, users.map((u) => u.cluster))) + const names = (list: typeof evidence) => list.map((e) => e.cluster.name).join(', ') + if (evidence.length === 0) return { text: 'Unknown', tone: 'unknown', evidence: summary.text } + if (summary.tone === 'unhealthy') { + const archiving = evidence.filter((e) => e.archiving.tone === 'unhealthy') + const backups = evidence.filter((e) => e.window?.failingSinceLastSuccess) + const parts = [ + archiving.length > 0 ? `WAL archiving failing on ${names(archiving)}` : null, + backups.length > 0 ? `a backup failed after the last success for ${names(backups)}` : null, + ].filter(Boolean) + return { text: summary.text, tone: summary.tone, evidence: `Inferred: ${parts.join('; ')}` } + } + const archiving = evidence.filter((e) => e.archiving.tone === 'healthy') + if (summary.tone === 'healthy') { + return { text: summary.text, tone: summary.tone, evidence: `Inferred from WAL archiving on ${names(archiving)}` } + } + return { + text: summary.text, + tone: summary.tone, + evidence: archiving.length > 0 + ? `Archiving on ${names(archiving)}; no archiving result from the others` + : 'Its clusters report no archiving result yet', + } +} + +export function CNPGProtection({ + data, + fleet, + namespaces, + searchParams, + onSetParams, + onInspect, + inspected, + onClearNamespaces, + scopeCluster, +}: CNPGScreenProps & { scopeCluster?: { namespace: string; name: string } }) { + const clusterFilter = scopeCluster ? `${scopeCluster.namespace}/${scopeCluster.name}` : searchParams.get('cluster') + const rows = useMemo( + () => fleet.rows.filter((r) => !clusterFilter || `${r.namespace}/${r.name}` === clusterFilter), + [fleet.rows, clusterFilter], + ) + + const failed = useMemo(() => { + const now = Date.now() + return (data.objects.backups ?? []) + .filter((b) => isApiGroup(b.apiVersion, 'postgresql.cnpg.io')) + .filter((b) => { + const level = getCNPGBackupStatus(b).level + if (level !== 'unhealthy' && level !== 'alert') return false + const t = Date.parse(backupTime(b) ?? '') + return Number.isFinite(t) && now - t <= WEEK_MS + }) + .filter((b) => !clusterFilter || `${b.metadata?.namespace}/${b.spec?.cluster?.name}` === clusterFilter) + .sort((a, b) => Date.parse(backupTime(b) ?? '') - Date.parse(backupTime(a) ?? '')) + }, [data.objects.backups, clusterFilter]) + + const stores = useMemo(() => { + return (data.objects.objectStores ?? []).map((s) => { + const ns = s.metadata?.namespace ?? '' + const name = s.metadata?.name ?? '' + const users = fleet.rows.filter( + (r) => r.namespace === ns && getCNPGClusterBarmanPlugin(r.cluster)?.barmanObjectName === name, + ) + return { key: `${ns}/${name}`, namespace: ns, name, destination: getCNPGObjectStoreDestination(s), users, health: storeHealth(s, users) } + }).filter((s) => !clusterFilter || s.users.some((u) => `${u.namespace}/${u.name}` === clusterFilter)) + }, [data.objects.objectStores, fleet.rows, clusterFilter]) + + const schedules = useMemo( + () => + (data.objects.scheduledBackups ?? []).filter( + (s) => !clusterFilter || `${s.metadata?.namespace}/${s.spec?.cluster?.name}` === clusterFilter, + ), + [data.objects.scheduledBackups, clusterFilter], + ) + + const chips = scopeCluster ? [] : [ + ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), + ...namespaceChip(namespaces, onClearNamespaces), + ] + const backupsReadable = data.coverage.backups?.state === 'full' + + return ( +
+ {!scopeCluster && ( + + )} + + + + + ( + <> +
{r.name}
+ {r.namespace} + + ), + }, + { header: 'Schedule', width: '13%', cell: (r) => }, + { + header: 'Last successful backup', + width: '16%', + cell: (r) => ( + <> + + {r.protection.lastSuccessfulBackup.source && {r.protection.lastSuccessfulBackup.source}} + + ), + }, + { header: 'WAL archiving', width: '16%', cell: (r) => }, + { + header: 'Recovery window', + width: '13%', + cell: (r) => + r.protection.recoveryWindow.from ? ( + <> + + from {ageText(r.protection.recoveryWindow.from)} + + + {r.protection.recoveryWindow.tone === 'degraded' ? 'not advancing: archiving failing' : 'to the newest archived WAL'} + + + ) : ( + + ), + }, + { + header: 'Restore validation', + width: '13%', + cell: (r) => ( + + ), + }, + { + header: 'Destination', + width: '15%', + cell: (r) => , + }, + ]} + rows={rows} + rowKey={(r) => r.key} + rowResource={(r) => clusterResource(r.namespace, r.name)} + onInspect={onInspect} + inspected={inspected} + minWidth={1000} + empty={coverageEmpty(data.coverage.clusters, 'PostgreSQL clusters')} + footer="Kubernetes records no restore tests, so restore validation is never shown as passed. Recovery windows come from ObjectStore status." + /> + + {b.metadata?.name} }, + { header: 'Cluster', width: '16%', cell: (b) => <>{b.spec?.cluster?.name ?? '—'}{b.metadata?.namespace} }, + { header: 'Started', width: '12%', cell: (b) => ageText(backupStart(b)) }, + { + header: 'Error', + width: '44%', + cell: (b) => {b.status?.error || getCNPGBackupStatus(b).text}, + }, + ]} + rows={failed} + rowKey={(b) => `${b.metadata?.namespace}/${b.metadata?.name}`} + rowResource={(b) => cnpgResource('backups', b.metadata?.namespace, b.metadata?.name)} + onInspect={onInspect} + inspected={inspected} + empty={backupsReadable && data.coverage.backups?.state === 'full' ? 'No failed backups in the last 7 days.' : coverageEmpty(data.coverage.backups, 'failed backups')} + /> + + <>{s.name}{s.namespace} }, + { header: 'Destination', width: '30%', cell: (s) => {s.destination} }, + { header: 'Used by', width: '20%', cell: (s) => (s.users.length ? s.users.map((u) => u.name).join(', ') : None visible) }, + { + header: 'Upload health (inferred)', + width: '32%', + cell: (s) => ( + <> + {s.health.text} + {s.health.evidence} + + ), + }, + ]} + rows={stores} + rowKey={(s) => s.key} + rowResource={(s) => cnpgResource('objectstores', s.namespace, s.name, 'barmancloud.cnpg.io')} + onInspect={onInspect} + inspected={inspected} + empty={data.coverage.objectStores?.state === 'notInstalled' ? 'The barman-cloud plugin’s ObjectStore kind is not installed.' : coverageEmpty(data.coverage.objectStores, 'ObjectStores')} + footer="ObjectStore has no health status of its own; upload health is inferred from its clusters’ WAL archiving and backup results." + /> + + <>{s.metadata?.name}{s.metadata?.namespace} }, + { header: 'Cluster', width: '16%', cell: (s) => s.spec?.cluster?.name ?? '—' }, + { + header: 'Schedule', + width: '24%', + cell: (s) => ( + <> + {s.spec?.schedule ?? '—'} + CNPG cron, seconds first + + ), + }, + { + header: 'Status', + width: '12%', + cell: (s) => { + const st = getCNPGScheduledBackupStatus(s) + return {st.text} + }, + }, + { header: 'Last run', width: '12%', cell: (s) => ageText(s.status?.lastScheduleTime) }, + { + header: 'Next run', + width: '12%', + cell: (s) => { + const next = s.status?.nextScheduleTime + if (!next) return '—' + const ms = Date.parse(next) - Date.now() + return ms >= 0 ? `in ${formatDuration(ms)}` : `${formatDuration(-ms)} overdue` + }, + }, + ]} + rows={schedules} + rowKey={(s) => `${s.metadata?.namespace}/${s.metadata?.name}`} + rowResource={(s) => cnpgResource('scheduledbackups', s.metadata?.namespace, s.metadata?.name)} + onInspect={onInspect} + inspected={inspected} + empty={coverageEmpty(data.coverage.scheduledBackups, 'ScheduledBackups')} + footer={data.backupsOmitted > 0 ? `${data.backupsOmitted} settled backups older than 7 days are not listed.` : undefined} + /> +
+
+ ) +} diff --git a/web/src/components/cnpg/CNPGSummaryHost.tsx b/web/src/components/cnpg/CNPGSummaryHost.tsx new file mode 100644 index 0000000000..43dcf92177 --- /dev/null +++ b/web/src/components/cnpg/CNPGSummaryHost.tsx @@ -0,0 +1,143 @@ +import type { ReactNode } from 'react' +import { useNavigate } from 'react-router-dom' +import { + CNPG_BARMAN_OBJECTSTORE_GROUP, + CNPG_GROUP, + CNPGBackupSummary, + CNPGClusterSummary, + CNPGDatabaseSummary, + CNPGImageCatalogSummary, + CNPGObjectStoreSummary, + CNPGPoolerSummary, + CNPGPublicationSummary, + CNPGScheduledBackupSummary, + CNPGSubscriptionSummary, + PaneLoader, + isApiGroup, + refToSelectedResource, + type CNPGNavigate, + type CNPGRef, + type CNPGWorkspaceResponse, + type NavigateToResource, +} from '@skyhook-io/k8s-ui' +import { useCNPGFleet } from './useCNPGSidebarWorkspace' +import { cnpgClusterFullPath, currentPageLabel } from './paths' +import { useConnection } from '../../context/ConnectionContext' + +interface SummaryContext { + apiKind: string + namespace: string + name: string + group?: string + resource: any + context: 'drawer' | 'expanded' + onNavigate?: NavigateToResource +} + +function ClusterSummaryHost({ namespace, name, context, onNavigate }: SummaryContext) { + const navigate = useNavigate() + const { connection } = useConnection() + // The workspace is read for the object's own namespace: an explicitly opened + // Cluster shows its facts whatever the namespace filter is. + const { query, fleet } = useCNPGFleet([namespace]) + const row = fleet?.rows.find((r) => r.namespace === namespace && r.name === name) + if (!row) { + if (query.isLoading) return + return ( +
+ {query.error instanceof Error + ? `The CloudNativePG summary could not be loaded: ${query.error.message}` + : 'This Cluster is not in the CloudNativePG workspace for your identity.'}{' '} + Spec & status still shows everything the object reports. +
+ ) + } + const go = onNavigate ? (ref: CNPGRef) => onNavigate(refToSelectedResource(ref)) : undefined + return ( + + navigate(cnpgClusterFullPath(namespace, name, connection.context || undefined), { + state: { returnLabel: currentPageLabel(), returnCtx: connection.context }, + }), + }, + { + label: 'Logs', + onClick: () => + navigate(cnpgClusterFullPath(namespace, name, connection.context || undefined, 'logs'), { + state: { returnLabel: currentPageLabel(), returnCtx: connection.context }, + }), + }, + { + label: 'Protection', + onClick: () => + navigate(cnpgClusterFullPath(namespace, name, connection.context || undefined, 'protection'), { + state: { returnLabel: currentPageLabel(), returnCtx: connection.context }, + }), + }, + ] + : undefined + } + /> + ) +} + +type ObjectSummary = (props: { resource: any; workspace: CNPGWorkspaceResponse | null; onNavigate?: CNPGNavigate }) => ReactNode + +const OBJECT_SUMMARIES: Record = { + Backup: CNPGBackupSummary, + ScheduledBackup: CNPGScheduledBackupSummary, + Pooler: CNPGPoolerSummary, + Database: CNPGDatabaseSummary, + Publication: CNPGPublicationSummary, + Subscription: CNPGSubscriptionSummary, + ImageCatalog: CNPGImageCatalogSummary, + ClusterImageCatalog: CNPGImageCatalogSummary, +} + +function ObjectSummaryHost({ ctx, Summary }: { ctx: SummaryContext; Summary: ObjectSummary }) { + // A ClusterImageCatalog is referenced from any namespace, so its users are + // read across every namespace the caller can see. + const clusterScoped = ctx.resource?.kind === 'ClusterImageCatalog' + const { query } = useCNPGFleet(clusterScoped ? [] : [ctx.namespace]) + if (query.isLoading) return + const workspace = query.data?.installed ? query.data : null + const go = ctx.onNavigate ? (ref: CNPGRef) => ctx.onNavigate?.(refToSelectedResource(ref)) : undefined + return +} + +// The object's own apiVersion decides: Velero also ships a Backup kind. +function groupOf(ctx: SummaryContext): string | undefined { + const apiVersion = ctx.resource?.apiVersion + if (typeof apiVersion !== 'string') return ctx.group + if (isApiGroup(apiVersion, CNPG_GROUP)) return CNPG_GROUP + if (isApiGroup(apiVersion, CNPG_BARMAN_OBJECTSTORE_GROUP)) return CNPG_BARMAN_OBJECTSTORE_GROUP + return undefined +} + +function renderSummaryFor(ctx: SummaryContext): ReactNode { + const group = groupOf(ctx) + const kind = ctx.resource?.kind + if (group === CNPG_BARMAN_OBJECTSTORE_GROUP && kind === 'ObjectStore') { + return + } + if (group !== CNPG_GROUP) return null + if (kind === 'Cluster') return + const Summary = OBJECT_SUMMARIES[kind] + return Summary ? : null +} + +/** + * The composed Overview for CloudNativePG kinds. Returns null for kinds + * without one, which keeps the default Overview. + */ +export function renderCNPGSummary(ctx: SummaryContext): ReactNode { + return renderSummaryFor(ctx) +} diff --git a/web/src/components/cnpg/CNPGView.tsx b/web/src/components/cnpg/CNPGView.tsx new file mode 100644 index 0000000000..0dc899d196 --- /dev/null +++ b/web/src/components/cnpg/CNPGView.tsx @@ -0,0 +1,156 @@ +import { useCallback, useEffect, useMemo, useRef } from 'react' +import { useLocation, useNavigate, useSearchParams } from 'react-router-dom' +import { ResourcesSidebar, type SelectedKindInfo } from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { useAPIResources } from '../../api/apiResources' +import { usePinnedKinds } from '../../hooks/useFavorites' +import { useResourceCounts } from '../../hooks/useResourceCounts' +import { CNPGOverview } from './CNPGOverview' +import { CNPGProtection } from './CNPGProtection' +import { CNPGDeclarations } from './CNPGDeclarations' +import { CNPGPooling } from './CNPGPooling' +import { CNPGOperator } from './CNPGOperator' +import { CNPGScreenGate } from './shared' +import { CNPGDetailPage } from './CNPGDetailPage' +import { decodeDrawerTrail, encodeDrawerTrail, parseCNPGRoute, sameResource } from './routes' +import { useCNPGFleet, useCNPGSidebarWorkspace } from './useCNPGSidebarWorkspace' + +interface CNPGViewProps { + namespaces: string[] + selectedResource: SelectedResource | null + onOpenResource: (resource: SelectedResource) => void + onCloseResource: () => void + onClearNamespaces: () => void +} + +/** + * The CloudNativePG workspace. Lives inside Resources (the rail keeps + * Resources highlighted) and renders the same Resources sidebar, with the + * workspace destinations above the exact CNPG kinds. + * + * The drawer is the app's single resource drawer. `?drawer=` backs it so a + * refresh, a shared link and browser Back restore the inspected object. + */ +export function CNPGView({ namespaces, selectedResource, onOpenResource, onCloseResource, onClearNamespaces }: CNPGViewProps) { + const location = useLocation() + const navigate = useNavigate() + const [searchParams, setSearchParams] = useSearchParams() + const route = parseCNPGRoute(location.pathname) + const { data: apiResources } = useAPIResources() + const { data: counts } = useResourceCounts(namespaces) + const { pinned, togglePin, isPinned } = usePinnedKinds() + const sidebarWorkspace = useCNPGSidebarWorkspace({ + apiResources, + namespaces, + active: { screen: route.screen, child: route.detail ? { label: route.detail.name, title: `${route.detail.plural} ${route.detail.namespace}/${route.detail.name}` } : undefined }, + }) + const { query, fleet } = useCNPGFleet(namespaces) + + const drawerParam = searchParams.get('drawer') + const trail = useMemo(() => decodeDrawerTrail(drawerParam), [drawerParam]) + const drawerTarget = trail.length > 0 ? trail[trail.length - 1] : null + + // Two-way sync between ?drawer= and the app drawer. Whichever side changed + // since the last sync wins, so URL navigation (Back, a pasted link) opens the + // drawer and drawer navigation (close, a link inside it) rewrites the URL. + const lastSynced = useRef(null) + const selectedKey = selectedResource ? encodeDrawerTrail([selectedResource]) : '' + const targetKey = drawerTarget ? encodeDrawerTrail([drawerTarget]) : '' + useEffect(() => { + if (targetKey !== (lastSynced.current ?? '')) { + lastSynced.current = targetKey + if (drawerTarget && !sameResource(drawerTarget, selectedResource)) onOpenResource(drawerTarget) + else if (!drawerTarget && selectedResource) onCloseResource() + return + } + if (selectedKey !== targetKey) { + lastSynced.current = selectedKey + const params = new URLSearchParams(searchParams) + if (!selectedResource) { + params.delete('drawer') + } else { + const idx = trail.findIndex((r) => sameResource(r, selectedResource)) + const next = idx >= 0 ? trail.slice(0, idx + 1) : [...trail, selectedResource] + params.set('drawer', encodeDrawerTrail(next)) + } + setSearchParams(params, { replace: true, state: location.state }) + } + }, [targetKey, selectedKey]) // eslint-disable-line react-hooks/exhaustive-deps + + const inspect = useCallback( + (resource: SelectedResource) => { + const params = new URLSearchParams(searchParams) + params.set('drawer', encodeDrawerTrail([resource])) + setSearchParams(params, { replace: true, state: location.state }) + }, + [searchParams, setSearchParams, location.state], + ) + + const setParams = useCallback( + (update: Record) => { + const params = new URLSearchParams(searchParams) + for (const [k, v] of Object.entries(update)) { + if (v === null || v === '') params.delete(k) + else params.set(k, v) + } + setSearchParams(params, { replace: true, state: location.state }) + }, + [searchParams, setSearchParams, location.state], + ) + + const selectKind = useCallback( + (kind: SelectedKindInfo) => { + navigate(`/resources/${kind.name}${kind.group ? `?apiGroup=${encodeURIComponent(kind.group)}` : ''}`) + }, + [navigate], + ) + + return ( +
+ isPinned(kind, group ?? '')} + categoryWorkspaces={sidebarWorkspace} + /> +
+ {route.detail ? ( + + ) : ( + + {(data, readyFleet) => { + const props = { + data, + fleet: readyFleet, + namespaces, + searchParams, + onSetParams: setParams, + onInspect: inspect, + inspected: drawerTarget, + onClearNamespaces, + } + switch (route.screen) { + case 'protection': + return + case 'declarations': + return + case 'pooling': + return + case 'operator': + return + default: + return + } + }} + + )} +
+
+ ) +} diff --git a/web/src/components/cnpg/paths.ts b/web/src/components/cnpg/paths.ts new file mode 100644 index 0000000000..a6f561ae66 --- /dev/null +++ b/web/src/components/cnpg/paths.ts @@ -0,0 +1,13 @@ +import { cnpgDetailPath } from './routes' + +export function cnpgClusterFullPath(namespace: string, name: string, ctx?: string, tab?: string): string { + return cnpgDetailPath({ plural: 'clusters', namespace, name }, ctx, tab) +} + +/** + * The label for "← back" on the page a push lands on: the title of the page + * being left, which Radar keeps in the document title. + */ +export function currentPageLabel(): string { + return document.title.replace(/\s*·\s*Radar$/, '') || 'previous page' +} diff --git a/web/src/components/cnpg/routes.test.ts b/web/src/components/cnpg/routes.test.ts new file mode 100644 index 0000000000..ea390980ab --- /dev/null +++ b/web/src/components/cnpg/routes.test.ts @@ -0,0 +1,60 @@ +import { describe, expect, it } from 'vitest' +import { cnpgDetailKindFor, cnpgDetailPath, decodeDrawerTrail, encodeDrawerTrail, parseCNPGRoute, sameResource } from './routes' + +describe('CNPG routes', () => { + it('parses workspace screens and falls back to Overview for unknown or unavailable ones', () => { + expect(parseCNPGRoute('/cnpg').screen).toBe('overview') + expect(parseCNPGRoute('/cnpg/').screen).toBe('overview') + expect(parseCNPGRoute('/cnpg/nope').screen).toBe('overview') + expect(parseCNPGRoute('/cnpg/protection').screen).toBe('protection') + expect(parseCNPGRoute('/cnpg/operator').screen).toBe('operator') + }) + + it('round-trips a drawer trail and keeps the API group', () => { + const trail = [ + { kind: 'backups', group: 'postgresql.cnpg.io', namespace: 'payments', name: 'pg-billing-20260928020000' }, + { kind: 'objectstores', group: 'barmancloud.cnpg.io', namespace: 'payments', name: 's3-billing' }, + { kind: 'clusterimagecatalogs', group: 'postgresql.cnpg.io', namespace: '', name: 'postgresql-standard' }, + ] + const encoded = encodeDrawerTrail(trail) + expect(decodeDrawerTrail(encoded)).toEqual(trail) + }) + + it('drops malformed entries instead of opening a guessed object', () => { + expect(decodeDrawerTrail('clusters:postgresql.cnpg.io:payments:pg-orders~garbage')).toEqual([ + { kind: 'clusters', group: 'postgresql.cnpg.io', namespace: 'payments', name: 'pg-orders' }, + ]) + expect(decodeDrawerTrail(null)).toEqual([]) + }) + + it('distinguishes same-named kinds from different groups', () => { + const cnpg = { kind: 'clusters', group: 'postgresql.cnpg.io', namespace: 'a', name: 'x' } + const capi = { kind: 'clusters', group: 'cluster.x-k8s.io', namespace: 'a', name: 'x' } + expect(sameResource(cnpg, capi)).toBe(false) + expect(sameResource(cnpg, { ...cnpg })).toBe(true) + }) + + it('parses full-detail routes for every CNPG kind and the cluster-scoped placeholder', () => { + expect(parseCNPGRoute('/cnpg/clusters/payments/pg-orders')).toEqual({ + screen: 'overview', + detail: { plural: 'clusters', group: 'postgresql.cnpg.io', namespace: 'payments', name: 'pg-orders' }, + }) + expect(parseCNPGRoute('/cnpg/objectstores/payments/s3-billing').screen).toBe('protection') + expect(parseCNPGRoute('/cnpg/objectstores/payments/s3-billing').detail?.group).toBe('barmancloud.cnpg.io') + expect(parseCNPGRoute('/cnpg/clusterimagecatalogs/_/postgresql-standard').detail?.namespace).toBe('') + expect(parseCNPGRoute('/cnpg/clusters/payments').detail).toBeUndefined() + }) + + it('builds detail paths carrying the context', () => { + expect(cnpgDetailPath({ plural: 'clusters', namespace: 'payments', name: 'pg-orders' }, 'prod', 'logs')).toBe( + '/cnpg/clusters/payments/pg-orders?ctx=prod&tab=logs', + ) + expect(cnpgDetailPath({ plural: 'clusterimagecatalogs', namespace: '', name: 'std' })).toBe('/cnpg/clusterimagecatalogs/_/std') + }) + + it('only claims CNPG kinds from the CNPG groups', () => { + expect(cnpgDetailKindFor('clusters', 'postgresql.cnpg.io')).toBe('clusters') + expect(cnpgDetailKindFor('clusters', 'cluster.x-k8s.io')).toBeNull() + expect(cnpgDetailKindFor('backups', 'velero.io')).toBeNull() + }) +}) diff --git a/web/src/components/cnpg/routes.ts b/web/src/components/cnpg/routes.ts new file mode 100644 index 0000000000..bd3cab25b9 --- /dev/null +++ b/web/src/components/cnpg/routes.ts @@ -0,0 +1,112 @@ +import type { SelectedResource } from '../../types' + +export type CNPGScreen = 'overview' | 'protection' | 'declarations' | 'pooling' | 'operator' + +export const CNPG_SCREENS: { id: CNPGScreen; label: string; path: string }[] = [ + { id: 'overview', label: 'Overview', path: '/cnpg' }, + { id: 'protection', label: 'Protection', path: '/cnpg/protection' }, + { id: 'declarations', label: 'Declarations', path: '/cnpg/declarations' }, + { id: 'pooling', label: 'Pooling', path: '/cnpg/pooling' }, + { id: 'operator', label: 'Operator', path: '/cnpg/operator' }, +] + +export interface CNPGDetailTarget { + plural: string + group: string + namespace: string + name: string +} + +export interface CNPGRoute { + screen: CNPGScreen + detail?: CNPGDetailTarget +} + +// The CNPG kinds that have a CNPG-framed full detail, with the workspace +// destination each one lives under. +export const CNPG_DETAIL_KINDS: Record = { + clusters: { group: 'postgresql.cnpg.io', kind: 'Cluster', home: 'overview' }, + backups: { group: 'postgresql.cnpg.io', kind: 'Backup', home: 'protection' }, + scheduledbackups: { group: 'postgresql.cnpg.io', kind: 'ScheduledBackup', home: 'protection' }, + objectstores: { group: 'barmancloud.cnpg.io', kind: 'ObjectStore', home: 'protection' }, + databases: { group: 'postgresql.cnpg.io', kind: 'Database', home: 'declarations' }, + publications: { group: 'postgresql.cnpg.io', kind: 'Publication', home: 'declarations' }, + subscriptions: { group: 'postgresql.cnpg.io', kind: 'Subscription', home: 'declarations' }, + poolers: { group: 'postgresql.cnpg.io', kind: 'Pooler', home: 'pooling' }, + imagecatalogs: { group: 'postgresql.cnpg.io', kind: 'ImageCatalog', home: 'operator' }, + clusterimagecatalogs: { group: 'postgresql.cnpg.io', kind: 'ClusterImageCatalog', home: 'operator', clusterScoped: true }, +} + +export function cnpgDetailKindFor(plural: string, group: string | undefined): string | null { + const p = plural.toLowerCase() + const spec = CNPG_DETAIL_KINDS[p] + return spec && spec.group === (group ?? '') ? p : null +} + +export function parseCNPGRoute(pathname: string): CNPGRoute { + const seg = pathname.replace(/^\/+/, '').split('/').map((s) => { + try { + return decodeURIComponent(s) + } catch { + return s + } + }) + if (seg[0] !== 'cnpg') return { screen: 'overview' } + const s = seg[1] ?? '' + const detailSpec = CNPG_DETAIL_KINDS[s] + if (detailSpec && seg[2] && seg[3]) { + return { + screen: detailSpec.home, + detail: { plural: s, group: detailSpec.group, namespace: seg[2] === '_' ? '' : seg[2], name: seg[3] }, + } + } + const match = CNPG_SCREENS.find((x) => x.id === s) + if (match) return { screen: match.id } + return { screen: 'overview' } +} + +/** Full detail path; `ctx` pins the Kubernetes context the object belongs to. */ +export function cnpgDetailPath(target: Omit, ctx?: string, tab?: string): string { + const params = new URLSearchParams() + if (ctx) params.set('ctx', ctx) + if (tab) params.set('tab', tab) + const qs = params.toString() + return `/cnpg/${target.plural}/${encodeURIComponent(target.namespace || '_')}/${encodeURIComponent(target.name)}${qs ? `?${qs}` : ''}` +} + +export function cnpgScreenPath(screen: CNPGScreen): string { + return CNPG_SCREENS.find((s) => s.id === screen)?.path ?? '/cnpg' +} + +// Drawer identity in the URL: kind:group:namespace:name, chained with "~" for +// the in-drawer trail (last entry is the one shown). Kubernetes names and API +// groups cannot contain ":" or "~", and the group is mandatory — CNPG's Cluster +// and Backup collide with CAPI, KubeBlocks and Velero kinds. +export function encodeDrawerRef(r: SelectedResource): string { + return [r.kind, r.group ?? '', r.namespace ?? '', r.name].join(':') +} + +export function decodeDrawerRef(s: string): SelectedResource | null { + const parts = s.split(':') + if (parts.length !== 4 || !parts[0] || !parts[3]) return null + return { kind: parts[0], group: parts[1], namespace: parts[2], name: parts[3] } +} + +export function decodeDrawerTrail(param: string | null): SelectedResource[] { + if (!param) return [] + return param.split('~').map(decodeDrawerRef).filter((r): r is SelectedResource => r !== null) +} + +export function encodeDrawerTrail(trail: SelectedResource[]): string { + return trail.map(encodeDrawerRef).join('~') +} + +export function sameResource(a: SelectedResource | null | undefined, b: SelectedResource | null | undefined): boolean { + if (!a || !b) return false + return ( + a.kind.toLowerCase() === b.kind.toLowerCase() && + (a.group ?? '') === (b.group ?? '') && + (a.namespace ?? '') === (b.namespace ?? '') && + a.name === b.name + ) +} diff --git a/web/src/components/cnpg/shared.tsx b/web/src/components/cnpg/shared.tsx new file mode 100644 index 0000000000..7a16081e59 --- /dev/null +++ b/web/src/components/cnpg/shared.tsx @@ -0,0 +1,290 @@ +import type { ReactNode } from 'react' +import type { UseQueryResult } from '@tanstack/react-query' +import { clsx } from 'clsx' +import { Database, X } from 'lucide-react' +import { + CNPG_KIND_BY_KEY, + PaneLoader, + type CNPGFleet, + type CNPGKindCoverage, + type CNPGWorkspaceResponse, +} from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { useConnection } from '../../context/ConnectionContext' +import { EmptyState, Notice, ROW_HOVER, TABLE_HEAD, TABLE_WRAP, TBODY, TD, TH } from '../capacity/shared' +import { sameResource } from './routes' + +export interface CNPGScreenProps { + data: CNPGWorkspaceResponse + fleet: CNPGFleet + namespaces: string[] + searchParams: URLSearchParams + onSetParams: (update: Record) => void + onInspect: (resource: SelectedResource) => void + inspected: SelectedResource | null + onClearNamespaces: () => void +} + +const COVERAGE_LABEL: Record = { + denied: 'no access', + partial: 'no access in some namespaces', + syncing: 'still loading', + error: 'could not be read', +} + +export function CNPGWorkspaceHeader({ title, subtitle, actions }: { title: string; subtitle?: ReactNode; actions?: ReactNode }) { + return ( +
+
+ + CloudNativePG +
+
+
+

{title}

+ {subtitle &&
{subtitle}
} +
+ {actions} +
+
+ ) +} + +export function CoverageNotice({ fleet, data }: { fleet: CNPGFleet; data: CNPGWorkspaceResponse }) { + if (fleet.incompleteKinds.length === 0) return null + const parts = fleet.incompleteKinds.map((k) => { + const cov = data.coverage[k] + const label = COVERAGE_LABEL[cov?.state ?? ''] ?? cov?.state + return `${CNPG_KIND_BY_KEY[k].kind} (${label})` + }) + return ( + + Some CloudNativePG data is not readable: {parts.join(', ')}. Facts built on it read “No access” or “unknown” rather than none, and counts are lower bounds. + + ) +} + +/** + * Loading, error and not-installed states every workspace screen shares. + * Renders the screen only once there is workspace data to render. + */ +export function CNPGScreenGate({ + query, + fleet, + children, +}: { + query: UseQueryResult + fleet: CNPGFleet | null + children: (data: CNPGWorkspaceResponse, fleet: CNPGFleet) => ReactNode +}) { + const { connection } = useConnection() + const data = query.data + if (!data && query.isLoading) return + if (!data) { + return ( + + ) + } + if (!data.installed || !fleet) { + return ( + + ) + } + return <>{children(data, fleet)} +} + +export function ScreenBody({ children }: { children: ReactNode }) { + return ( +
+
{children}
+
+ ) +} + +export function FilterChips({ chips }: { chips: { label: string; onClear: () => void }[] }) { + if (chips.length === 0) return null + return ( +
+ {chips.map((c) => ( + + {c.label} + + + ))} +
+ ) +} + +export function namespaceChip(namespaces: string[], onClear: () => void) { + return namespaces.length > 0 ? [{ label: `Namespace: ${namespaces.join(', ')}`, onClear }] : [] +} + +export function Segments({ + value, + options, + onChange, + label, +}: { + value: T + options: { id: T; label: string; count?: number }[] + onChange: (id: T) => void + label: string +}) { + return ( +
+ {options.map((o) => { + const on = o.id === value + return ( + + ) + })} +
+ ) +} + +export interface TableColumn { + header: ReactNode + width?: string + cell: (row: T) => ReactNode + className?: string +} + +/** A workspace table. Rows inspect in the drawer; the inspected row is highlighted. */ +export function SectionTable({ + title, + subtitle, + columns, + rows, + rowKey, + rowResource, + onInspect, + inspected, + empty, + minWidth = 760, + footer, +}: { + title: ReactNode + subtitle?: ReactNode + columns: TableColumn[] + rows: T[] + rowKey: (row: T) => string + rowResource?: (row: T) => SelectedResource | null + onInspect?: (resource: SelectedResource) => void + inspected?: SelectedResource | null + empty: ReactNode + minWidth?: number + footer?: ReactNode +}) { + return ( +
+
+

{title}

+ {subtitle && {subtitle}} +
+
+ {rows.length === 0 ? ( +
{empty}
+ ) : ( +
+ + + {columns.map((c, i) => ( + + ))} + + + + {columns.map((c, i) => ( + + ))} + + + + {rows.map((row) => { + const res = rowResource?.(row) ?? null + const active = !!res && sameResource(inspected, res) + return ( + onInspect(res) : undefined} + className={clsx(res && onInspect && 'cursor-pointer', ROW_HOVER, active && 'selection')} + aria-selected={res ? active : undefined} + > + {columns.map((c, i) => ( + + ))} + + ) + })} + +
{c.header}
{c.cell(row)}
+
+ )} +
+ {footer &&
{footer}
} +
+ ) +} + +/** Empty-state text for a collection, derived from how much of it was readable. */ +export function coverageEmpty(cov: CNPGKindCoverage | undefined, noun: string): string { + switch (cov?.state) { + case 'full': + return `No ${noun} in this scope.` + case 'partial': + return `No ${noun} visible. Some namespaces are not readable with your access.` + case 'denied': + return `No access to ${noun}.` + case 'syncing': + return `${noun[0].toUpperCase()}${noun.slice(1)} are still loading.` + case 'error': + return `${noun[0].toUpperCase()}${noun.slice(1)} could not be read.` + default: + return `This kind is not installed.` + } +} + +/** The less complete of two coverages, for collections built from several kinds. */ +export function worstCoverage(...covs: (CNPGKindCoverage | undefined)[]): CNPGKindCoverage | undefined { + const rank: Record = { error: 0, denied: 1, syncing: 2, partial: 3, full: 4, notInstalled: 5 } + return covs.filter(Boolean).sort((a, b) => (rank[a!.state] ?? 9) - (rank[b!.state] ?? 9))[0] +} + +export function Mono({ children }: { children: ReactNode }) { + return {children} +} + +export function Sub({ children }: { children: ReactNode }) { + return
{children}
+} + +export function clusterResource(namespace: string, name: string): SelectedResource { + return { kind: 'clusters', group: 'postgresql.cnpg.io', namespace, name } +} + +export function cnpgResource(plural: string, namespace: string, name: string, group = 'postgresql.cnpg.io'): SelectedResource { + return { kind: plural, group, namespace, name } +} diff --git a/web/src/components/cnpg/useCNPGSidebarWorkspace.ts b/web/src/components/cnpg/useCNPGSidebarWorkspace.ts new file mode 100644 index 0000000000..d963a5698e --- /dev/null +++ b/web/src/components/cnpg/useCNPGSidebarWorkspace.ts @@ -0,0 +1,88 @@ +import { useMemo } from 'react' +import { useNavigate } from 'react-router-dom' +import { Database, FileCheck2, Settings2, ShieldCheck, Waypoints } from 'lucide-react' +import { buildCNPGFleet, type CNPGFleet, type SidebarCategoryWorkspace } from '@skyhook-io/k8s-ui' +import type { APIResource } from '../../types' +import { useCNPGWorkspace } from '../../api/cnpg' +import { CNPG_SCREENS, type CNPGScreen } from './routes' + +export const CNPG_SIDEBAR_CATEGORY = 'CloudNativePG' + +const ICONS: Record = { + overview: Database, + protection: ShieldCheck, + declarations: FileCheck2, + pooling: Waypoints, + operator: Settings2, +} + +export function cnpgDiscovered(apiResources: APIResource[] | undefined): boolean { + return !!apiResources?.some((r) => r.group === 'postgresql.cnpg.io') +} + +export function useCNPGFleet(namespaces: string[], enabled = true) { + const query = useCNPGWorkspace(namespaces, { enabled }) + const fleet = useMemo(() => (query.data?.installed ? buildCNPGFleet(query.data) : null), [query.data]) + return { query, fleet } +} + +function destinationCount(screen: CNPGScreen, fleet: CNPGFleet | null): { count?: number | null; title?: string } { + if (!fleet) return {} + const lowerBound = fleet.incompleteKinds.length > 0 ? ' Some CloudNativePG data is not readable, so this is a lower bound.' : '' + switch (screen) { + case 'overview': + return { count: fleet.attentionCount, title: `${fleet.attentionCount} clusters need attention.${lowerBound}` } + case 'protection': + return { count: fleet.categoryCounts.protection, title: `${fleet.categoryCounts.protection} clusters with failing backups or WAL archiving.${lowerBound}` } + case 'declarations': + return { count: fleet.categoryCounts.declarations, title: `${fleet.categoryCounts.declarations} clusters with declarations that are not reconciled.${lowerBound}` } + case 'pooling': + return { count: fleet.categoryCounts.pooling, title: `${fleet.categoryCounts.pooling} clusters with Pooler problems.${lowerBound}` } + default: + return {} + } +} + +/** + * The CloudNativePG workspace entries for the Resources sidebar. Returns + * undefined when CNPG is not discovered, so clusters without the operator see + * an unchanged sidebar. + */ +export function useCNPGSidebarWorkspace({ + apiResources, + namespaces, + active, +}: { + apiResources: APIResource[] | undefined + namespaces: string[] + active?: { screen: CNPGScreen; child?: { label: string; title?: string } } +}): Record | undefined { + const navigate = useNavigate() + const discovered = cnpgDiscovered(apiResources) + const { fleet } = useCNPGFleet(namespaces, discovered) + const nsKey = namespaces.join(',') + + return useMemo(() => { + if (!discovered) return undefined + const destinations = CNPG_SCREENS.map((s) => { + const { count, title } = destinationCount(s.id, fleet) + return { + id: s.id, + label: s.label, + icon: ICONS[s.id], + count, + countTitle: title, + active: active?.screen === s.id, + child: active?.screen === s.id ? active.child : undefined, + onSelect: () => navigate(s.path), + } + }) + return { + [CNPG_SIDEBAR_CATEGORY]: { + destinations, + defaultKindsCollapsed: !!active, + scopeNote: nsKey ? `Counts for namespace ${nsKey.split(',').join(', ')}` : undefined, + }, + } + }, [discovered, fleet, active?.screen, active?.child?.label, active?.child?.title, navigate, nsKey]) // eslint-disable-line react-hooks/exhaustive-deps +} diff --git a/web/src/components/resources/ResourceDetailDrawer.tsx b/web/src/components/resources/ResourceDetailDrawer.tsx index d4f778be74..b6e43822bd 100644 --- a/web/src/components/resources/ResourceDetailDrawer.tsx +++ b/web/src/components/resources/ResourceDetailDrawer.tsx @@ -1,6 +1,7 @@ import { ResourceDetailDrawer as BaseResourceDetailDrawer } from '@skyhook-io/k8s-ui' import type { SelectedResource } from '../../types' import { WorkloadView } from '../workload/WorkloadView' +import { CNPGDrawerTrailBack } from '../cnpg/CNPGDrawerTrail' interface ResourceDetailDrawerProps { resource: SelectedResource @@ -31,6 +32,9 @@ export function ResourceDetailDrawer(props: ResourceDetailDrawerProps) { return ( {({ resource, expanded, active, initialTab, onClose, onExpand, onExpandIntent, onCancelExpandIntent, onBack, onNavigateToResource, onCollapseToDrawer }) => ( +
+ {!expanded && } +
+
+
)}
) diff --git a/web/src/components/resources/ResourcesView.tsx b/web/src/components/resources/ResourcesView.tsx index 5b65b571c9..981969cbaf 100644 --- a/web/src/components/resources/ResourcesView.tsx +++ b/web/src/components/resources/ResourcesView.tsx @@ -10,6 +10,8 @@ import { useAPIResources } from '../../api/apiResources' import { useConnection } from '../../context/ConnectionContext' import { initNavigationMap, getSecretStoreProviderType } from '@skyhook-io/k8s-ui' import { usePinnedKinds } from '../../hooks/useFavorites' +import { useResourceCounts } from '../../hooks/useResourceCounts' +import { useCNPGSidebarWorkspace } from '../cnpg/useCNPGSidebarWorkspace' import { useOpenLogs, useOpenWorkloadLogs } from '../dock' import { canBulkRestartKind, @@ -26,13 +28,6 @@ import { apiVersionToGroup, kindToPluralWithGroup, type NavigateToResource } fro import { CreateResourceDialog } from '../shared/CreateResourceDialog' import { getSkeletonYaml } from '../../utils/skeleton-yaml' -interface ResourceCountsResponse { - counts: Record - forbidden?: string[] - reasons?: Record - unavailable?: string[] -} - interface ResourcesViewProps { namespaces: string[] selectedResource?: SelectedResource | null @@ -114,6 +109,8 @@ export function ResourcesView({ namespaces, selectedResource, onResourceClick, o if (apiResources) initNavigationMap(apiResources) }, [apiResources]) + const cnpgSidebarWorkspace = useCNPGSidebarWorkspace({ apiResources, namespaces }) + // Track the selected kind from the k8s-ui component const [selectedKind, setSelectedKind] = useState(null) const workloadWrites = namespaces.length === 0 @@ -126,37 +123,7 @@ export function ResourcesView({ namespaces, selectedResource, onResourceClick, o // Lightweight resource counts for sidebar badges (~2KB instead of ~608MB) const namespacesParam = namespaces.join(',') - const { data: countsData, isError: countsIsError } = useQuery({ - queryKey: ['resource-counts', namespacesParam], - queryFn: async () => { - const params = new URLSearchParams() - if (namespaces.length > 0) params.set('namespaces', namespacesParam) - const startedAt = performance.now() - debugNamespaceLog('resources:counts-fetch-start', { namespaces, params: params.toString() }) - try { - return await fetchJSON(`/resource-counts?${params}`) - } finally { - debugNamespaceLog('resources:counts-fetch-end', { - namespaces, - params: params.toString(), - durationMs: Math.round(performance.now() - startedAt), - }) - } - }, - staleTime: 10000, - // SSE invalidation isn't running while connecting, and mid-sync counts - // are what unlatch guarded kinds as their informers finish — poll fast - // during the shell, settle to the safety net once connected. - refetchInterval: connection.state === 'connecting' ? 3000 : 60000, - // During the first seconds of the progressive shell the endpoint 503s - // (cluster_connecting) until the mid-sync cache handle exists; keep the - // query pending rather than parking it in error state, which would - // unlatch the large-list guard at the connected flip. - retry: (failureCount: number, error: Error) => - isStillLoadingError(error) ? true : failureCount < 3, - retryDelay: (failureCount: number, error: Error) => - isStillLoadingError(error) ? 2000 : Math.min(1000 * 2 ** failureCount, 30000), - }) + const { data: countsData, isError: countsIsError } = useResourceCounts(namespaces) // Determine if selected kind is a CRD (only CRDs should send ?group= to backend) const isSelectedCrd = useMemo(() => { @@ -440,6 +407,7 @@ export function ResourcesView({ namespaces, selectedResource, onResourceClick, o connectionState={connection.state === 'connecting' && connection.syncStatus ? 'syncing' : connection.state} largeListGuard={largeListGuard} onSelectedKindChange={setSelectedKind} + sidebarCategoryWorkspaces={cnpgSidebarWorkspace} topPodMetrics={topPodMetrics} topNodeMetrics={topNodeMetrics} certExpiry={certExpiry} diff --git a/web/src/components/workload/WorkloadView.tsx b/web/src/components/workload/WorkloadView.tsx index d41d80418c..4bf37bb4da 100644 --- a/web/src/components/workload/WorkloadView.tsx +++ b/web/src/components/workload/WorkloadView.tsx @@ -3,18 +3,20 @@ import { JobRenderer, JobSetRenderer } from '../resources/renderers/JobAdmission import { RayClusterRenderer } from '../resources/renderers/RayClusterRenderer' import { RayServiceRenderer } from '../resources/renderers/RayServiceRenderer' import { KueueWorkloadRenderer } from '../resources/renderers/KueueWorkloadRenderer' -import { useMemo, useEffect, useCallback, useRef, useState } from 'react' +import { useMemo, useEffect, useCallback, useRef, useState, type ReactNode } from 'react' import { useQueries, useQueryClient } from '@tanstack/react-query' -import { useNavigate, useLocation, useSearchParams } from 'react-router-dom' +import { Navigate, useNavigate, useLocation, useSearchParams } from 'react-router-dom' import { workloadPodAwaitsScheduling } from '../capacity/podDemandGate' import { clsx } from 'clsx' import { Terminal, Stethoscope } from 'lucide-react' import { WorkloadView as BaseWorkloadView, + isApiGroup, EditableYamlView, FetchResult, Section, type WorkloadTabType, + type WorkloadExtraTab, type RendererOverrides, type GitOpsOwnerRef, type GitOpsStatus, @@ -148,6 +150,9 @@ import { CNPGSubscriptionRenderer, } from '../resources/renderers/CNPGDeclarativeRenderer' import { CreateResourceDialog } from '../shared/CreateResourceDialog' +import { renderCNPGSummary } from '../cnpg/CNPGSummaryHost' +import { CNPGClusterLogs } from '../cnpg/CNPGClusterLogs' +import { cnpgDetailKindFor, cnpgDetailPath } from '../cnpg/routes' import { cleanYamlForDuplicate } from '../../utils/skeleton-yaml' import { useDesktopDownload } from '../../hooks/useDesktopDownload' import { useCompareLauncher } from '../compare/useCompareLauncher' @@ -277,6 +282,17 @@ export function WorkloadViewRoute({ onNavigateToResource }: WorkloadViewRoutePro ) } + const cnpgPlural = cnpgDetailKindFor(kind, group) + if (cnpgPlural) { + const params = new URLSearchParams(searchParams) + params.delete('apiGroup') + const tab = params.get('tab') + if (cnpgPlural === 'clusters' && (tab === 'timeline' || tab === 'events')) params.set('tab', 'activity') + const base = cnpgDetailPath({ plural: cnpgPlural, namespace, name }) + const qs = params.toString() + return + } + return ( )} + renderSummary={({ apiKind: ak, namespace: ns, name: n, resource: res, context, onNavigate }) => + renderCNPGSummary({ apiKind: ak, namespace: ns, name: n, group: effectiveGroup, resource: res, context, onNavigate }) + } renderExpandedOverview={({ kind: k, apiKind, namespace: ns, name: n, resource: res }) => supportsBatchExecution(k, apiKind, effectiveGroup, res?.apiVersion) && res ? ( @@ -1672,6 +1694,10 @@ function LogsTabContent({ ) } + if (kind === 'Cluster' && isApiGroup(resource?.apiVersion, 'postgresql.cnpg.io')) { + return + } + // Workload kinds with stable pod selectors use the aggregated workload logs viewer if (WORKLOAD_LOG_KINDS.has(kind) && (kind !== 'Job' || isCoreBatchJob(apiKind, group))) { return ( diff --git a/web/src/hooks/useResourceCounts.ts b/web/src/hooks/useResourceCounts.ts new file mode 100644 index 0000000000..c1b1396fbb --- /dev/null +++ b/web/src/hooks/useResourceCounts.ts @@ -0,0 +1,49 @@ +import { useQuery } from '@tanstack/react-query' +import { debugNamespaceLog, fetchJSON, isStillLoadingError } from '../api/client' +import { useConnection } from '../context/ConnectionContext' + +export interface ResourceCountsResponse { + counts: Record + forbidden?: string[] + reasons?: Record + unavailable?: string[] +} + +// Lightweight per-kind counts for the Resources sidebar badges. Shared by the +// Resources view and any surface that renders the same sidebar standalone, so +// both read one cache entry. +export function useResourceCounts(namespaces: string[]) { + const { connection } = useConnection() + const namespacesParam = namespaces.join(',') + return useQuery({ + queryKey: ['resource-counts', namespacesParam], + queryFn: async () => { + const params = new URLSearchParams() + if (namespaces.length > 0) params.set('namespaces', namespacesParam) + const startedAt = performance.now() + debugNamespaceLog('resources:counts-fetch-start', { namespaces, params: params.toString() }) + try { + return await fetchJSON(`/resource-counts?${params}`) + } finally { + debugNamespaceLog('resources:counts-fetch-end', { + namespaces, + params: params.toString(), + durationMs: Math.round(performance.now() - startedAt), + }) + } + }, + staleTime: 10000, + // SSE invalidation isn't running while connecting, and mid-sync counts + // are what unlatch guarded kinds as their informers finish — poll fast + // during the shell, settle to the safety net once connected. + refetchInterval: connection.state === 'connecting' ? 3000 : 60000, + // During the first seconds of the progressive shell the endpoint 503s + // (cluster_connecting) until the mid-sync cache handle exists; keep the + // query pending rather than parking it in error state, which would + // unlatch the large-list guard at the connected flip. + retry: (failureCount: number, error: Error) => + isStillLoadingError(error) ? true : failureCount < 3, + retryDelay: (failureCount: number, error: Error) => + isStillLoadingError(error) ? 2000 : Math.min(1000 * 2 ** failureCount, 30000), + }) +}