From 0c80dce80a704ee5d8c7199a3364fef9e4dc4ee6 Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Tue, 29 Sep 2026 02:07:13 +0300 Subject: [PATCH 01/17] Add CloudNativePG workspace: fleet Overview, sidebar destinations and composed Cluster summary - GET /api/cnpg/workspace: every CNPG kind plus owner-validated instance Pods, authorized per kind (namespaced kinds fall back per namespace; ClusterImageCatalog needs a cluster-scope list), with coverage states, CNPG issues from the Issues engine and the no-schedule audit finding withheld where coverage is missing. - k8s-ui: buildCNPGFleet derives per-cluster facts (instances, replication, protection as separate schedule/destination/last-backup/WAL/recovery-window/ restore-validation facts, declarations) without inventing values: replication lag is unknown without runtime data, restore validation is never green, and unreadable kinds read "No access" rather than none. - ResourcesSidebar gains optional category workspaces (destinations above a collapsible Resource kinds block, grouped by API group). - WorkloadView gains an optional composed summary: Overview shows it and the kind's renderer moves to a Spec & status tab. - /cnpg: Overview fleet with Needs attention / All, problem-category chips, search, namespace chip and a URL-backed drawer (?drawer=kind:group:ns:name). --- CLAUDE.md | 1 + docs/plans/CNPG_WORKSPACE.md | 168 +++++ internal/server/cnpg_workspace.go | 615 +++++++++++++++++ internal/server/cnpg_workspace_test.go | 464 +++++++++++++ internal/server/server.go | 1 + .../components/cnpg/CNPGClusterSummary.tsx | 220 ++++++ packages/k8s-ui/src/components/cnpg/index.ts | 3 + .../k8s-ui/src/components/cnpg/primitives.tsx | 131 ++++ .../src/components/cnpg/workspace.test.ts | 179 +++++ .../k8s-ui/src/components/cnpg/workspace.ts | 646 ++++++++++++++++++ .../resources/ResourcesSidebar.test.tsx | 76 +++ .../components/resources/ResourcesSidebar.tsx | 194 +++++- .../components/resources/ResourcesView.tsx | 6 +- .../k8s-ui/src/components/resources/index.ts | 2 +- .../src/components/workload/WorkloadView.tsx | 61 +- packages/k8s-ui/src/index.ts | 3 + web/src/App.tsx | 28 +- web/src/api/cnpg.ts | 21 + web/src/components/cnpg/CNPGOverview.tsx | 374 ++++++++++ web/src/components/cnpg/CNPGSummaryHost.tsx | 52 ++ web/src/components/cnpg/CNPGView.tsx | 126 ++++ web/src/components/cnpg/paths.ts | 5 + web/src/components/cnpg/routes.test.ts | 34 + web/src/components/cnpg/routes.ts | 64 ++ .../cnpg/useCNPGSidebarWorkspace.ts | 88 +++ .../components/resources/ResourcesView.tsx | 44 +- web/src/components/workload/WorkloadView.tsx | 4 + web/src/hooks/useResourceCounts.ts | 49 ++ 28 files changed, 3601 insertions(+), 58 deletions(-) create mode 100644 docs/plans/CNPG_WORKSPACE.md create mode 100644 internal/server/cnpg_workspace.go create mode 100644 internal/server/cnpg_workspace_test.go create mode 100644 packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/index.ts create mode 100644 packages/k8s-ui/src/components/cnpg/primitives.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/workspace.test.ts create mode 100644 packages/k8s-ui/src/components/cnpg/workspace.ts create mode 100644 web/src/api/cnpg.ts create mode 100644 web/src/components/cnpg/CNPGOverview.tsx create mode 100644 web/src/components/cnpg/CNPGSummaryHost.tsx create mode 100644 web/src/components/cnpg/CNPGView.tsx create mode 100644 web/src/components/cnpg/paths.ts create mode 100644 web/src/components/cnpg/routes.test.ts create mode 100644 web/src/components/cnpg/routes.ts create mode 100644 web/src/components/cnpg/useCNPGSidebarWorkspace.ts create mode 100644 web/src/hooks/useResourceCounts.ts diff --git a/CLAUDE.md b/CLAUDE.md index 3ef78afee..280cb2ed4 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -182,6 +182,7 @@ After `make -demo`, run `kubectl config use-context kind-radar--demo - RBAC reverse-lookup: `/api/rbac/subject/{kind}/{namespace}/{name}` (ServiceAccount) and `/api/rbac/subject/{kind}/{name}` (User/Group) return direct + group-inherited bindings + flattened effective rules. SA subjects also get a `usedByPods` list (Pods whose `spec.serviceAccountName` matches — closes the loop on the SA detail page). `/api/rbac/role/{kind}/{namespace}/{name}` (use `_` for ClusterRole's empty namespace) returns the inverse — bindings that reference the role + their subjects. `/api/rbac/namespace/{namespace}` returns RoleBindings in the namespace + ClusterRoleBindings with at least one SA subject in it + a ServiceAccount count (backs the NamespaceRenderer's RBAC section; group-only ClusterRoleBindings like `system:authenticated` grants are deliberately excluded — they'd appear in every namespace and would be noise). `/api/rbac/whoami?namespace=...` is a pass-through of `SelfSubjectRulesReview` for the current user. Backed by `pkg/rbac/` (pure index + 5s TTL memo); endpoints gate on `list rolebindings` AND `list clusterrolebindings` (403 when either is denied — silent partial views would mislead operators). - Policy (Kyverno) reverse-lookup: `/api/policy/resource/{kind}/{ns}/{name}` returns one resource's policy findings; `/api/policy/policies/{policy}?namespace=&limit=` returns the inverse — every resource one policy recorded an outcome for, per rule. Report families are authorized **per subject scope** (`policyreports` cluster-wide is a different grant from `clusterpolicyreports`), findings from an unreadable family are dropped from lists AND counts with the withheld count reported, and `counts` describe the cluster while subject lists are capped and follow the namespace view filter — the two must be read together. `/api/policy/policies/{policy}/queued` returns the policy's in-flight `UpdateRequest`s; Kyverno records those in its own namespace, so it reads cluster-wide gated on `list updaterequests` rather than inheriting the caller's view filter. - Velero reverse-lookup: `/api/velero/backupstoragelocations/{ns}/{name}/backups` returns the Backups a storage location holds, with each one's phase, completion time and expiration, plus how many reached `Completed`. Gated on `list backups`; an unset `spec.storageLocation` resolves to whichever location carries `spec.default`, falling back to the name `default` when none does. Backs the storage location's "Stored Here" section, which states what an `Unavailable` location is holding back — the Backup's own status cannot see its location's health. `POST /api/velero/{backups|restores}/{ns}/{name}/messages` returns the warnings and errors behind a run's counts: it creates a `DownloadRequest`, waits for Velero to answer with a pre-signed URL, and fetches the results file from object storage. Impersonated (it creates a CR); needs a running Velero controller and object storage reachable from wherever Radar runs, and reports which of the two failed rather than returning an empty list. +- CloudNativePG workspace: `/api/cnpg/workspace` returns every CNPG kind (plus owner-validated instance Pods) with per-kind `coverage` (`full|partial|denied|notInstalled|syncing|error`, `partial` naming only in-scope denied namespaces), CNPG issues from the Issues engine and `cnpgNoDeclarativeBackup` audit findings, each withheld where the caller lacks coverage. Namespaced kinds follow the view filter and the capacity per-namespace `list` fallback; `ClusterImageCatalog` needs a cluster-scope `list`. Handler `internal/server/cnpg_workspace.go` - CloudNativePG reverse-lookup: `/api/cnpg/imagecatalogs/{ns}/{name}/clusters` and `/api/cnpg/clusterimagecatalogs/{name}/clusters` return the Clusters pinned to an image catalog, with the major each asks for and the image it actually resolved. Cluster-scoped catalogs are referenceable from any namespace, so the cluster-scoped route reads cluster-wide gated on `list clusters` — a view-filtered answer would report "nothing uses this" before an edit. ## Key Patterns diff --git a/docs/plans/CNPG_WORKSPACE.md b/docs/plans/CNPG_WORKSPACE.md new file mode 100644 index 000000000..a4727191b --- /dev/null +++ b/docs/plans/CNPG_WORKSPACE.md @@ -0,0 +1,168 @@ +# CloudNativePG Workspace — implementation plan + +Source design: Claude Design project `6656e6ed-4e96-41b9-84d2-427d8e9c42fa` (`CNPG Workspace.dc.html`, `CNPG IA Notes.dc.html`). Prototype data is synthetic; this plan maps each screen onto what a real cluster exposes and says where the design must bend. + +Status: DRAFT (revised after Codex cycle 1) — awaiting sign-off. Branch `feature/cnpg-workspace`. Delivered as 6 stacked PRs with a checkpoint after PR3. + +--- + +## 0. Premise check (read first) + +The IA is sound and fits Radar: CNPG is already a first-class integration (10 kinds registered `internal/k8s/dynamic_cache.go:302-313`, renderers for every kind, issues `internal/issues/source_cnpg.go`, audit `pkg/audit/cnpg.go`, demo `scripts/cnpg-demo`). What's missing is a *task-shaped* surface across those kinds. The workspace is the right next step. But five parts of the prototype assume data or mechanisms that don't exist, and building them as drawn would violate the certainty contract: + +| Prototype element | Reality | Plan | +|---|---|---| +| **Runtime · Sessions & locks** (per-pid table with query text, blocking pid) | Instance manager `/pg/status` has no sessions/locks (CNPG `pkg/postgres/status.go`). Per-session rows need `psql` via **pods/exec**. Metrics give only aggregates (`cnpg_backends_total`, `cnpg_backends_waiting_total`, `cnpg_backends_max_tx_duration_seconds`). | v1 ships **aggregates from Prometheus** (sessions by state, waiting-on-locks count, oldest tx age). Per-session table = Needs-input Q1 (exec is a big privilege step). | +| **Runtime denied** message says "denied pods/exec" | Replication/slots/WAL come from `/pg/status` via **pods/proxy** (what `kubectl cnpg status` does: `ProxyGet(scheme,pod,"8000","/pg/status")`). | Denied state names `get pods/proxy` (and, separately, Prometheus unavailability). | +| **Recovery verified** ("Restore drill 6 d ago") | Kubernetes records no restore drills. A Ready Cluster bootstrapped via `bootstrap.recovery` proves a recovery *happened once*, not that today's archive is recoverable. | Column renamed **"Restore evidence"**, **never green**. Default "Not observed" (unknown-grey). When a Cluster in this context has `bootstrap.recovery` sourcing this store/serverName: neutral "Restored into pg-x · created " with that provenance. Needs-input Q2. | +| **URLs carry the Kubernetes context** (`/c/prod-us-east1/cnpg/...`) | Radar's context is server state, not URL (`useSwitchContext` `web/src/api/client.ts:6527`); on switch App keeps pathname **and `location.state`** and drops params (`web/src/App.tsx:1181-1233`). The dangerous case is not a 404 — it's a **same-name object in the new context** silently rendering a different database. | App-wide context-in-URL stays out of scope. CNPG detail routes (and the drawer param) carry **`ctx=`**. If it differs from the active context: render "pg-orders is not in " with **Switch back to ** and **Go to Overview**; never fetch the same-name object. Collections have no `ctx` and simply re-query. `onContextChanged` clears all params today (`web/src/App.tsx:1221-1228`), so it gains a CNPG exception that **preserves `ctx`** on `/cnpg/clusters/*` and `?drawer=` routes (like the existing `openAfter` exception, :1203-1212); a link **without** `ctx` (pasted from outside) is treated as the active context. Tested through the real switch callback. Every write (§2.6) also sends `reviewedContext` + object `uid`. | +| **Return stack** as its own app state | Radar navigation is URL + browser history; no lineage beyond GitOps `?from=` (single level, `GitOpsView.tsx:367-384`). | Implement "← previous task" as **browser history + `location.state.returnLabel`**: every CNPG navigation pushes with a label of the origin; the control reads it and calls `navigate(-1)`. Fresh tab ⇒ no state ⇒ control hidden (IA Q7 for free). Filters/tab/drawer are already in the URL, so Back restores them. No parallel stack. | + +Also note: the Pooling "pressure" and all trend sparklines require Prometheus scraping CNPG's `:9187` / `:9127`. Without series they render "No metrics found" — never zero, and never "PodMonitor not enabled" unless `spec.monitoring.enablePodMonitor` is explicitly false (other scrape setups exist). + +**Premise risk (surfaced, not resolved — Q6):** the strongest journey is "find the unhealthy cluster and understand its availability/protection problem". A smaller design — Overview fleet + Protection feeding the existing detail surfaces — serves it. The heavier parts (Runtime, Actions) carry most of the risk and complexity. The user chose full scope; the phasing below front-loads the fleet journeys and puts an explicit **checkpoint after PR3** before Runtime/Actions. + +## 1. Resolved IA questions (from the notes) + +1. Sidebar highlight on detail opened from kind list → destination-based highlight; return label carries the origin. **Adopt.** +2. Managed role detail home → Cluster **Configuration › Managed roles** anchor. **Adopt.** +3. Which navigations push history → destination changes push; tab/filter/drawer changes `replace`. **Adopt** (matches existing `?full=1` push / collapse replace, `App.tsx:831-836`). +4. ObjectStore ownership → stays in Protection; health always labelled *inferred*, evidence listed per user cluster. **Adopt** (data: `status.serverRecoveryWindow` keyed by serverName + each user Cluster's `ContinuousArchiving` / `LastBackupSucceeded`; the existing host wrapper already derives `clusterForServer`/`archivingFailing`, `web/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx`). +5. Pooling destination vs section → keep both. **Adopt.** +6. Namespace filter vs open object → collections only; an explicitly opened object stays with an "outside the namespace filter" note. **Adopt.** +7. Back on fresh tab → hide control when no return state. **Adopt** (falls out of §0 design). + +Naming: Declarations (not "Databases"), Protection, Pooling, Operator, Overview — adopt the notes' first-use analysis. + +## 2. Architecture + +### 2.1 Routing & nav — `/cnpg/*` view, Resources rail highlighted +- New `ExtendedMainView` `'cnpg'` in `web/src/App.tsx` (`getViewFromPath` :129-148 plus `CRASH_LABELS` :165, `radarPageTitle` :228, `namespaceFilterDisabled` :185 — **enabled** here: collections honour it). +- Routes: `/cnpg` (Overview), `/cnpg/protection`, `/cnpg/declarations`, `/cnpg/pooling`, `/cnpg/operator`, `/cnpg/clusters/:ns/:name/:tab?`. Filters in query (`?filter=attention&cat=&q=&cluster=&drawer=Kind|ns|name`). Router pattern mirrors Capacity (`parseCapacityRoute`, `web/src/components/capacity/shared.tsx:82-106`). +- Primary rail: `PrimaryNavRail` highlights **Resources** for `cnpg` (no new rail item — design invariant). +- Sidebar: `ResourcesSidebar` (k8s-ui) gets an optional **`categoryLinks?: Record`** prop (`{id,label,icon,count?,tone?,active,onSelect}` + optional nested `child`), rendered above a category's kinds under a "Workspace" label, with the kinds collapsed under "Resource kinds". Additive + optional ⇒ no break for Hub. The CloudNativePG category already exists (both groups map to it, `packages/k8s-ui/src/utils/api-resources.ts:242-243`). On `/cnpg/*` the web app renders `ResourcesSidebar` standalone (the package `ResourcesView` supports `hideSidebar`, `packages/k8s-ui/src/components/resources/ResourcesView.tsx:3412`; the web wrapper needs the sidebar's data props lifted into a shared hook so both surfaces feed it identically). +- Workspace problem counts: from the same `/api/cnpg/workspace` query that powers Overview (§2.3) so the sidebar and page never disagree. +- **Navigation model — exactly two mechanisms, never mixed:** + 1. *Browser history* for everything page-level. A `useCNPGNavigate()` wrapper: `push` sets `state.returnLabel` = label of the **current** entry (so the label always describes the entry `navigate(-1)` lands on); `replace` (tab, filter, search, drawer open/close) **copies the existing state forward**. "← {returnLabel}" renders iff `location.state?.returnLabel` exists and calls `navigate(-1)`. Refresh and duplicated tabs keep `history.state`, so the label stays truthful; fresh tabs have none ⇒ hidden. + 2. *Drawer trail* (`?drawer=` list) for in-drawer hops only; its "← Backup x" pops the trail (replace). **Expand** pushes the full page with `returnLabel` = the page under the drawer (the trail is not carried into the page). + - Context switch: App already clears params; CNPG pages also ignore a `returnLabel` whose entry carried a different `ctx`. + - Transition table (Overview→drawer→Backup→ObjectStore→Expand→Back; Resources›Cluster→Expand→Back; fresh-tab deep link; context switch on detail; namespace change with drawer open) is encoded as tests in PR1/PR2. + +### 2.2 Drawer — reuse the global drawer, add a back chain +- Rows open the existing global drawer (`navigateToResource`, `App.tsx:734-752`; shell `packages/k8s-ui/src/components/workload/ResourceDetailDrawer.tsx`). On `/cnpg/*` drawer state is URL-backed via `?drawer=` (same approach as `/resources?resource=`). +- **In-drawer back chain:** add `drawerTrail` to the drawer shell (k8s-ui): links inside the drawer push `{kind,ns,name}` onto the trail and show "← ". Chain lives in the `?drawer=` param as a `~`-joined list of **`Kind.group|ns|name`** refs (group mandatory — CNPG `Cluster`/`Backup` collide with CAPI/KubeBlocks/Velero, which the demo seeds on purpose) so Back/Forward and refresh work. Additive prop. +- **Expand**: for Cluster → navigate to `/cnpg/clusters/ns/name` (full page, §2.4). For other kinds → existing `?full=1` expanded WorkloadView. +- **Focused header** above existing renderers: new k8s-ui components `CNPGSubjectHeader` (problem banner + facts grid + relationship chips + actions) per kind, wired through the existing `rendererOverrides` host wrappers (`web/src/components/workload/WorkloadView.tsx:175-207`) so it appears in drawer, `/resources` drawer and full page alike. Renderers stay unchanged. Contract: the header shows **issues from the Issues engine** (top one + "+N more" → `/issues` filtered to the object) with a next action; the renderer's banners remain the detailed explanation. The header never re-derives health itself (no second precedence rule). + +### 2.3 Data — one gated aggregate endpoint; issues from the Issues engine; display derivation in TS +Why not pure client composition: `/api/resources/{kind}` does **not** per-kind gate namespaced CRDs — `preflightResourceList` only SARs Secrets and LimitRanges ("Other namespaced kinds are deferred", `internal/server/server.go:2139-2172`) and REST callers turn denials into `200 []`. The workspace can't make "No access" vs "none" claims on top of that. And the Issues engine (`internal/issues/source_cnpg.go`, API `pkg/issuesapi/types.go:364`) already detects CNPG problems with concurrent causes preserved; a TS re-derivation would be a divergent second problem system (the parity test only covers phase-string sets, `internal/issues/source_cnpg_parity_test.go`). + +- **`GET /api/cnpg/workspace`** (PR1): for each of the 10 CNPG kinds + instance Pods (`cnpg.io/cluster` label, owner-ref-validated against the Cluster) decide scope per kind **by discovery scope**: namespaced kinds use the capacity pattern (`internal/server/capacity_auth.go:28`: cluster-wide `canRead(list)` else `filterNamespacesByCanRead` over the user's namespaces) intersected with the namespace view filter; **cluster-scoped `ClusterImageCatalog` requires a cluster-scoped `list` SAR, no namespace fallback and no view-filter intersection**; read from the dynamic cache (`listDynamicSynced`), and return `{objects: {kind: [...] (summary-stripped)}, coverage: {kind: full|partial{deniedNamespaces}|denied|notInstalled|syncing}, issues: [CNPG issues for visible objects], audit: [cnpgNoDeclarativeBackup findings], operatorVersion}`. Not-installed ⇒ 200 with `installed:false`. +- **Pure derivation** in k8s-ui `utils/cnpg-workspace.ts`: `buildCNPGFleet(resp) → {rows, counts, categories}`. Unit-tested with fixtures from prototype scenarios + demo shapes. +- **Problems & counts:** "Needs attention" = clusters with ≥1 issue at warning+ attributed to the Cluster or to an object that references it (Backup/ScheduledBackup/Pooler/Database/… via `spec.cluster.name`). Category chips count **affected clusters**, not findings. Row shows top issue + "+N". Neutral **"No declarative backup schedule"** comes from the audit finding `cnpgNoDeclarativeBackup` — worded exactly to its meaning (`pkg/audit/cnpg.go:13`: no ScheduledBackup targets the Cluster; destination/on-demand backups may still exist). "No destination configured" is a separate derived fact from §Protection.2. "Runtime unavailable" category only exists once PR4 lands. +- **Per-cluster facts (only what is observed):** + - Instances: ready/total, primary (`currentPrimary`), designated-primary for replica clusters (`spec.replica.enabled` → "Replica cluster of "). Pills from pod readiness + role label. + - **Replication: "Unknown" without runtime.** Pod readiness and `instancesReportedState` (isPrimary/timeLineID/IP) do not establish streaming. With PR4 runtime: state + lag from `/pg/status`. + - **Protection is four separate facts**, each with its source: + 1. *Schedule*: active / suspended / none (ScheduledBackup `spec.suspend`). + 2. *Destination*: plugin ObjectStore / in-tree `barmanObjectStore` / volumeSnapshot / none (`getCNPGClusterBackupConfig`, `resource-utils-cnpg.ts:457`). + 3. *Last successful backup*: newest of {completed Backup CRs visible, ObjectStore `serverRecoveryWindow[serverName].lastSuccessfulBackupTime`, in-tree `status.lastSuccessfulBackup` when not plugin}, shown with which source won ("from ObjectStore status" / "from Backup pg-x"). Absent everywhere ⇒ "None observed", not "None". + 4. *WAL archiving*: `ContinuousArchiving` condition (absent ⇒ unknown). + Plus recovery window from ObjectStore and "Restore evidence" (§0). + - Declarations: `status.applied` (absent ⇒ Pending), managed roles from `managedRolesStatus.byStatus` and `cannotReconcile`. +- **Other backend additions:** + - `GET /api/cnpg/operator` (PR2) — operator + plugin discovery (Deployments labelled `app.kubernetes.io/name=cloudnative-pg`, plugin Services labelled `cnpg.io/pluginName`; config ConfigMap/Secret **names only**, ConfigMap data only with `get configmaps` in that ns; never Secret values). Each object read gated. + - `GET /api/cnpg/clusters/{ns}/{name}/logs` (+`/stream`) (PR3) — template `internal/server/jobset_logs.go:12`, `collectLogsFromPods(...,bounded=true)`, pods by label **and** controller ownerRef = this Cluster uid, gate `authorizePodLogRead`. Keeps the viewer's `{pod, container, timestamp, content, sourceLabel}` contract; adds optional parsed `level`/`logger` per entry from CNPG's JSON lines (raw passthrough otherwise). + - `GET /api/cnpg/clusters/{ns}/{name}/activity` (PR3) — §2.4. + - `GET /api/cnpg/clusters/{ns}/{name}/status` (PR4) — §2.5. + - `GET .../capabilities` + `POST .../{action}` (PR5) — §2.6. + +### 2.4 Cluster full page — `/cnpg/clusters/:ns/:name/:tab` +New `CNPGClusterPage` on `DetailShell` (`packages/k8s-ui/src/components/shared/DetailShell.tsx`; supply `breadcrumb` **and** `hideBackButton` — breadcrumb alone renders alongside the back nav). Header: return control (§0) · crumb "CloudNativePG / name" · name + `` + status · actions. +Tabs: +- **Overview**: focused problem + "Instances and replication" table + Related resources chips (Poolers, declared DBs, ScheduledBackup, ObjectStore, Pods, catalog, rw/ro/r Services, GitOps source from `argocd.argoproj.io/instance` / Flux labels) + existing `CNPGClusterRenderer` below (with `declared` slot, already wired). +- **Protection**: the three facts, backups table for this cluster, schedule, destination, recovery window; "Backup now" / "Restore to new cluster…". +- **Runtime**: §2.5. +- **Logs**: merged viewer — reuse k8s-ui `WorkloadLogsViewer` (`fetchAll → {pods, logs[{pod, sourceLabel, container, timestamp, content}]}`, `WorkloadLogsViewer.tsx:18`). Needs two **additive** props: `initialPod` and `levelFilter` (reads the optional parsed level). `?pod=` preselects (fleet terminal icon, "Open instance logs"). +- **Activity**: server-side `/api/cnpg/clusters/{ns}/{name}/activity` queries the timeline store (as `handleChanges`, `server.go:3840`, group-aware) for **all CNPG groups in the namespace** plus instance Pods, and keeps events whose object is the Cluster or references it (`spec.cluster.name`, `cnpg.io/cluster` label, owner) — so **deleted** Backups/declarations still appear. Verified gap: `TimelineEvent` stores no `spec.cluster.name` and `ExtractLabels` drops `cnpg.io/cluster` (`pkg/timeline/types.go:107`, `pkg/timeline/converter.go:197`). PR3 therefore **persists the relationship at ingestion**: for `postgresql.cnpg.io` / `barmancloud.cnpg.io` objects, the converter records the referenced cluster (`spec.cluster.name`, or `cnpg.io/cluster` label) as a retained label key — no storage schema change. No name-prefix matching. Events recorded before the upgrade lack it; the tab states "Child-object history is complete since ". Deduped, bounded, one cursor. +- **Configuration**: read-only structured spec: PostgreSQL parameters, storage, bootstrap, managed roles (anchor `#managed-roles`), plugins, monitoring, affinity. Links to YAML. +- **YAML**: existing `EditableYamlView`. +Namespace-filter note when the cluster is outside the current filter. + +### 2.5 Runtime (PR4) — two honest sources, each with its own unavailable state +0. **Security gate before any code (PR4 step 1):** `internal/server/curl.go:126-132` asserts apiserver `services/proxy` forwards the caller's credentials to the workload. Upstream apiserver deletes `Authorization` after successful authentication and strips `Impersonate-*`, so I expect pods/proxy to forward neither — but PR4 **proves it** with a demo-cluster echo pod recording received headers under both kubeconfig-token and in-cluster SA auth. If anything credential-bearing arrives, pods/proxy is dropped and runtime is Prometheus-only. +1. **Instance manager** via impersonated `pods/proxy` GET to `https::8000/pg/status` (scheme per pod: `--status-port-tls` in container command, as `remote.GetStatusSchemeFromPod`). Client = `getClientForRequest(r)` (`server.go:5227`) so apiserver enforces caller RBAC; pre-gate with `canReadSubresource(r,"","pods","proxy",ns,"get")` for a capability flag. Denial detection mirrors `pkg/probe/probe.go:720`. Yields: per-instance role, LSNs, WAL receiver, `replicationInfo[]` lags (`writeLag/flushLag/replayLag`, `syncState`), `replicationSlotsInfo[]`, archiver (`lastArchivedWAL[Time]`, `lastFailedWAL[Time]`, `readyWalFiles`), `pendingRestart`. Fan-out bounded (≤ instances, 5s timeout each), per-instance error preserved (one unreachable replica ≠ whole tab down). Response strips the embedded `pod` object. +2. **Prometheus** (optional, via `prometheuspkg.GetClient()`; gated with `canRead(clusters get)` in ns). **Isolation contract:** every query is built server-side (no user PromQL), selects `namespace=""` **and** `pod=~""` (regex-escaped), and applies whatever cluster-identity label Radar's curated resource charts already use for shared multi-cluster Prometheus (reuse that helper; don't invent one); duplicate scrapes collapsed with `max by (pod, …)`. Series: sessions by state (`cnpg_backends_total`), waiting (`cnpg_backends_waiting_total`), oldest tx (`cnpg_backends_max_tx_duration_seconds`), xid age, DB sizes (`cnpg_pg_database_size_bytes`), WAL/archiver rates, 1h trends (`QueryRange`), pooler `cnpg_pgbouncer_pools_cl_waiting/sv_active` — **pooler pods are selected separately**: Pooler → its generated Deployment (`resource-utils-cnpg.ts:1004`) → owned ReplicaSets → Pods, each hop from the cache with ownerRef validation, gated on `get poolers` in ns, same isolation rules. Absent Prometheus or series ⇒ "No metrics (Prometheus not connected / PodMonitor not enabled)". +Sub-tabs: Replication · Sessions · Transactions · Storage & WAL · Slots · Trends. Every value carries its source and sample time ("from instance manager · 5 s ago", "Prometheus · 30 s"). Three distinct states per source, never merged: **denied** (SAR false or apiserver Forbidden on proxy → names the grant), **unreachable** (permission OK, pod didn't answer — per instance), **absent** (no Prometheus / no series). Proxy denial with working metrics shows metrics; the design's dashed "Runtime data unavailable" card appears only when both sources are unavailable, listing what's still available. Fleet "Runtime unavailable" category = runtime capability false. + +### 2.6 Actions (PR5) — impersonated, SAR-gated, confirmed +Capabilities endpoint pattern from `handleRolloutCapabilities` (`internal/server/rollouts_handlers.go:106`), operations from `handleRolloutOperation` (:50) with `auth.AuditLog`, `getDynamicClientForRequest` (nil ⇒ 503, never SA fallback), error mapping like `writeRolloutError` (:224). Every POST carries `reviewedContext` + Cluster `uid` + `resourceVersion` seen in the confirm dialog; server rejects (409) on context/uid mismatch and **re-validates preconditions server-side** (not terminating, not hibernated, no switchover in flight, target is a current ready non-fenced instance of *this* Cluster). UI distinguishes *requested* (patch accepted) from *completed* (observed in status: `currentPrimary == target`, hibernation condition, Backup phase). + +| Action | Mechanism (as `kubectl cnpg` does) | SAR | +|---|---|---| +| Backup now | create `Backup` `{cluster, method}` where method ∈ the cluster's **configured** methods (plugin → `pluginConfiguration.name: barman-cloud.cloudnative-pg.io`; `barmanObjectStore`; `volumeSnapshot`) — dialog picks when >1; `target` default; name `-`; label `cnpg.io/cluster` | `create backups` | +| Switchover | status merge-patch `targetPrimary`, `targetPrimaryTimestamp`, `phase="Switchover in progress"`, `phaseReason` **with `metadata.resourceVersion` in the patch** (optimistic lock, as `kubectl cnpg promote`). On conflict: re-read, re-validate preconditions, and return 409 "cluster changed since you confirmed" — **no blind retry**. (Rollouts' `patchStatusThenSpec`, `pkg/rollouts/rollouts.go:220`, has no lock — shape reference only.) | `patch clusters/status` (separately from `patch clusters`) | +| Restart | annotation `kubectl.kubernetes.io/restartedAt` | `patch clusters` | +| Reload config | annotation `cnpg.io/reloadedAt` | `patch clusters` | +| Hibernate / Rehydrate | annotation `cnpg.io/hibernation=on/off`; progress from condition `cnpg.io/hibernation` | `patch clusters` | +| Restore to new cluster | **no bespoke endpoint**: dialog builds a new Cluster manifest — plugin: `bootstrap.recovery.source: origin` + `externalClusters: [{name: origin, plugin: {name: barman-cloud.cloudnative-pg.io, parameters: {barmanObjectName, serverName}}}]`; in-tree: `externalClusters[].barmanObjectStore` copied from source; or `recovery.backup.name` (same ns). Same image/catalog major as source, storage copied, optional `recoveryTarget.targetTime` bounded by the observed recovery window. Opens in the existing review flow: `/api/resources/preview` then `/api/resources/apply?mode=create&reviewedContext=` (strict create — a name collision fails, never updates, `server.go:4131-4191`). Dialog states that dry-run validates admission only, not archive reachability; new cluster has no WAL archiving unless added (warned, pointing at a different serverName). | apiserver-enforced | +| Edit YAML | existing editor | existing | + +Confirm dialogs name the effect ("Switchover promotes pg-orders-2; pg-orders-1 restarts as replica; writes pause for seconds"). Switchover target list excludes non-ready/fenced instances; disabled with reason when the cluster is mid-switchover/hibernated/terminating. Mutation toasts via React Query `meta` (CLAUDE.md frontend rule). + +## 3. Phased delivery (each PR independently shippable) + +| PR | Scope | Backend | Size | +|---|---|---|---| +| **1 Foundation + Overview** | `/cnpg` view + routes; sidebar `categoryLinks` (k8s-ui); `useCNPGWorkspace` + `buildCNPGFleet` (+ tests); Overview fleet (segments Needs attention/All, category chips, search, ns chip, issue column, instance pills; logs icon hidden until PR3); Cluster focused header in drawer; empty / not-installed / partial-coverage states; `useCNPGNavigate` + transition tests; `ctx` guard | `/api/cnpg/workspace` | L | +| **2 Protection · Declarations · Pooling · Operator + supporting headers** | four screens; composed Overview summaries (`CNPGSummary`) + Overview · Spec & status · YAML tabs for all supporting kinds, drawer and `/cnpg/:kind/...` full detail (§7); drawer back chain | `/api/cnpg/operator` | L | +| **3 Cluster full page** | tabs Overview (with K8s-only replication topology), Protection, Logs, Activity, Configuration, YAML; Expand → page; Resources-nav flyout on detail < 1600 px; not-in-context state; ns-filter note | merged logs (+stream) | L | +| — | **Checkpoint with user** after PR3: is Runtime/Actions still wanted as designed? (Q6) | | | +| **4 Runtime** | step 1 proxy-header proof (§2.5.0); demo gains Prometheus + PodMonitor fixture; Runtime tab; fleet lag/runtime category; pooler pressure | `/status` (pods/proxy) + Prometheus queries | M-L | +| **5 Actions** | demo gains a MinIO-backed `live-backup` mode (successful backup + restore; README constraints respected); capabilities + actions + Restore dialog via preview/apply | capabilities + action endpoints | M | +| **6 Docs + polish** | `docs/cnpg.md` (certainty contract like `docs/capacity.md`), `docs/integrations.md`, CLAUDE.md reference-docs row + endpoints list, README row fix (`README.md:549` stale). Docs for each endpoint land with its PR; this PR consolidates. | — | S | + +MCP: no new tools in this effort (the workspace is UI composition over data MCP already exposes). Revisit after. + +## 4. Testing +- k8s-ui unit: `buildCNPGFleet` fixtures (healthy, WAL failing, backup failed, declaration failed, forbidden kind, no backups configured, runtime unknown); sidebar `categoryLinks` render; header per kind. +- Go: handler tests using `newAuthTestServer`/`dynamicfake` patterns (`internal/server/server_auth_test.go:289`, `velero_handlers_test.go:328`): logs gate 403, status proxy denial → `runtime.denied`, capability SAR matrix (patch clusters vs clusters/status), action error mapping, no SA fallback when impersonation nil. Keep `cnpg_handlers_test.go` source-grep contracts intact. +- Coverage matrix tests for `/api/cnpg/workspace`: per-kind cluster-wide vs namespaced vs denied, colliding groups (Velero Backup, KubeBlocks Cluster from the demo), not installed. +- Context tests: same-name Cluster in two contexts → detail shows "not in this context", no fetch; write with stale `reviewedContext`/uid → 409. +- `/visual-test` on `make cnpg-demo` per PR with UI (frozen for states; `live` + new Prometheus / MinIO modes for PR4/5). +- `make tsc`, `make test`. + +## 5. Risks +- **Sidebar prop in k8s-ui** is public surface for Hub — additive only; Hub unaffected until it opts in. +- **Fleet payload size** on large fleets (design's 48-cluster variant): `/api/cnpg/workspace` returns summary-stripped objects; Backups are the long tail — cap to last 7 days + newest completed per cluster, with a count of omitted. +- **pods/proxy forwards caller credentials to the pod** (`internal/server/curl.go:126-132` caveat). Instance manager `/pg/status` is unauthenticated and read-only, so acceptable; we only ever GET `/pg/status`. +- **CNPG version drift**: phases matched on English sentences (demo README warning); `/pg/status` field names verified at v1.27 and v1.30. +- **Issues attribution** of child-object issues to clusters depends on `spec.cluster.name`; ObjectStore issues map to clusters via plugin `barmanObjectName`/`serverName` (same logic the ObjectStore host wrapper uses). + +## 6. Needs-input (blocking sign-off) +- **Q1 Sessions table**: ship aggregates only (Prometheus) — recommended — or add per-session rows via `pods/exec psql` (needs `create pods/exec`, shows query text = potential PII)? +- **Q2 Restore evidence**: keep the column as "Restore evidence" (never green; neutral provenance when a restored cluster exists) — recommended — or drop it? +- **Q3 Write actions scope** (RBAC posture): all of Backup now, Switchover, Restart, Reload, Hibernate/Rehydrate, Restore-via-apply — or a subset for v1? Should they be hidden behind an existing read-only/`--no-actions` setting if Radar has one? +- **Q4 Context guard**: `ctx=` only on CNPG detail/drawer URLs (recommended) vs app-wide context-in-URL as a separate project first? +- **Q5 PR cadence**: 6 stacked PRs off `feature/cnpg-workspace` (recommended) vs one long-lived branch merged at the end. +- **Q6 Premise checkpoint**: agree to pause after PR3 (fleet + Protection + Declarations + Pooling + Operator + Cluster page with Logs/Activity) to decide on Runtime/Actions with real usage in hand? + +## 7. Design revision 2 (Sep 29) — deltas folded in + +Revision 2 responds to `uploads/PROTOTYPE-REVIEW.md`; the design project also now carries `uploads/PLAN.md` (the IA decision doc) and `uploads/DESIGN-PROMPT.md`. The v1 file is kept as `CNPG Workspace v1.dc.html`. + +| Rev 2 change | Effect on this plan | +|---|---| +| **One composed detail**: CNPG drawers/pages have tabs **Overview · Spec & status · YAML**. Overview = composed fact hierarchy (e.g. Backup: Outcome / Relationships; ObjectStore: Upload health · inferred / Recovery window / Destination / Used by; Database: Declared / Reconciled / Source and target). The existing renderer moves to **Spec & status**, nothing repeats. | Replaces §2.2's "focused header above the existing renderer". Each CNPG kind gets a `CNPGSummary` (k8s-ui, pure) for Overview; existing renderers unchanged under Spec & status. Resolves Codex #13 duplication. | +| **Every CNPG kind has a CNPG full detail** (`uploads/PLAN.md`: the 10 kinds are "CNPG detail destinations"; expanding from Resources enters it, return goes back to Resources). | §2.2 "Expand for other kinds → `?full=1`" is dropped: route `/cnpg/:kind/:ns/:name/:tab` (`_` ns for ClusterImageCatalog) for all 10 kinds; Cluster keeps its 7 tabs. Resources drawer Expand on a CNPG kind navigates there with `returnLabel`. | +| **Replication topology** (`CNPGReplication`): primary card → replica cards with LSN, timeline, sync mode, lag pill + 60 s lag bar, Pooler chips in front; logs/pod links per instance. With runtime denied: "Topology from Kubernetes status. Lag and LSN need runtime access." | Matches §2.3 (replication Unknown without runtime). Topology from K8s (roles), LSN/lag/sync from `/pg/status` (PR4). Component lands in PR3 in its K8s-only form. | +| **Trends**: y-axis scale, threshold line, 15 min / 1 h range, hover values, hatched **gaps with reason** (not zero), per-chart source + sample coverage; **click a bar → that instance's logs ±6 min** as a removable chip. "Latest log lines" under replication. | PR4: Prometheus `QueryRange`; gaps = missing samples. Logs endpoint needs `sinceTime` + client-side upper bound for the ±6 min window (K8s logs API has no until). | +| **Scoped counts**: sidebar badges and kind counts follow the namespace filter with a "Counts for namespace …" note; Protection count = clusters with failing or unconfigured protection everywhere. | Already one source (`/api/cnpg/workspace`). Kind inventory counts come from `/resource-counts` which is already namespace-scoped. "Unconfigured" must use our wording ("no declarative schedule" / "no destination"), not "No backups". | +| **Qualified claims**: "Restore validation: None recorded"; ObjectStore "Uploads failing (inferred)" with source + time; recovery window "per Cluster status at 14:19"; controller phase labelled "reported by CNPG; Radar findings are separate"; ObjectStore Spec tab explains why "Recoverable" coexists with failing uploads. | Adopt the wording. **Two divergences kept:** (a) the prototype still shows one *green* "Restore drill 6 d ago" linking to Activity — Kubernetes records no drills, so we stay never-green (Q2); (b) recovery window "per Cluster status" — for plugin clusters `status.firstRecoverabilityPoint` is deprecated/unset, so we source it from ObjectStore `serverRecoveryWindow` and cite that. | +| **Correlation hints**: ObjectStore problem cites "Secret s3-billing-creds changed 2 minutes earlier"; Database problem explains "owner not among managed roles". | Both derivable (timeline event on the referenced credentials Secret; `spec.owner` vs `spec.managed.roles`). Shown as adjacent facts, not asserted causes ("Secret changed at 09:03" — no "because"). Roles may exist outside `managed.roles`, so the Database hint only appears when the operator error names the role. | +| **Return = drilldown only**: sidebar and global-nav hops clear the return stack. | Matches §2.1: only drilldown pushes set `returnLabel`; sidebar hops push without it ⇒ control hidden. Add to transition tests. | +| **Row actions**: labelled "Logs" (file icon, not terminal) + "Open →"; row inspects; no instruction sentence. | PR1 (Logs button appears once PR3 lands). | +| **Layout**: icon rail below 1600 px; on detail screens below 1600 px the Resources nav moves into a ☰ flyout beside the crumb; related resources stack under evidence below 1080 px main column; Runtime sub-sections are one segmented row. | Radar's rail already has an unpinned icon mode (`web/src/components/nav/PrimaryNavRail.tsx:48`). New work: sidebar flyout on CNPG detail routes (PR3). | +| **Kinds collapsed on workspace/detail**, open on Resources lists; user toggle wins. | `categoryLinks` prop gains `defaultKindsCollapsed`. | + +Still open from the design side: its notes say DESIGN.md / PLAN.md / the review "have not been attached", but they are now in `uploads/` — rev 2 may predate reading them, so a DESIGN.md conformance pass remains our job during implementation. diff --git a/internal/server/cnpg_workspace.go b/internal/server/cnpg_workspace.go new file mode 100644 index 000000000..58ff2e7db --- /dev/null +++ b/internal/server/cnpg_workspace.go @@ -0,0 +1,615 @@ +package server + +import ( + "context" + "errors" + "log" + "net/http" + "slices" + "sort" + "time" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/apimachinery/pkg/types" + + "github.com/skyhook-io/radar/internal/issues" + "github.com/skyhook-io/radar/internal/k8s" + bp "github.com/skyhook-io/radar/pkg/audit" + "github.com/skyhook-io/radar/pkg/issuesapi" +) + +const cnpgBarmanGroup = "barmancloud.cnpg.io" + +const ( + cnpgCoverageFull = "full" + cnpgCoveragePartial = "partial" + cnpgCoverageDenied = "denied" + cnpgCoverageNotInstalled = "notInstalled" + cnpgCoverageSyncing = "syncing" + cnpgCoverageError = "error" +) + +const ( + cnpgWorkspacePodsKey = "pods" + cnpgWorkspaceBackupsKey = "backups" + cnpgWorkspaceSchedKey = "scheduledBackups" + cnpgWorkspaceClusterKey = "clusters" +) + +// cnpgBackupWindow bounds how far back settled Backups are returned. The newest +// completed Backup per Cluster is kept regardless: it is the last-good-backup +// fact the workspace reports. +const cnpgBackupWindow = 7 * 24 * time.Hour + +const cnpgNoDeclarativeBackupCheckID = "cnpgNoDeclarativeBackup" + +type cnpgWorkspaceKind struct { + key string + group string + kind string + resource string + clusterScoped bool +} + +var cnpgWorkspaceKinds = []cnpgWorkspaceKind{ + {key: cnpgWorkspaceClusterKey, group: cnpgGroup, kind: "Cluster", resource: "clusters"}, + {key: cnpgWorkspaceBackupsKey, group: cnpgGroup, kind: "Backup", resource: "backups"}, + {key: cnpgWorkspaceSchedKey, group: cnpgGroup, kind: "ScheduledBackup", resource: "scheduledbackups"}, + {key: "poolers", group: cnpgGroup, kind: "Pooler", resource: "poolers"}, + {key: "databases", group: cnpgGroup, kind: "Database", resource: "databases"}, + {key: "publications", group: cnpgGroup, kind: "Publication", resource: "publications"}, + {key: "subscriptions", group: cnpgGroup, kind: "Subscription", resource: "subscriptions"}, + {key: "imageCatalogs", group: cnpgGroup, kind: "ImageCatalog", resource: "imagecatalogs"}, + {key: "clusterImageCatalogs", group: cnpgGroup, kind: "ClusterImageCatalog", resource: "clusterimagecatalogs", clusterScoped: true}, + {key: "objectStores", group: cnpgBarmanGroup, kind: "ObjectStore", resource: "objectstores"}, +} + +// CNPGWorkspaceCoverage states how much of one kind the caller could see. +// DeniedNamespaces lists only namespaces already in the caller's scope. +type CNPGWorkspaceCoverage struct { + State string `json:"state"` + DeniedNamespaces []string `json:"deniedNamespaces,omitempty"` +} + +// CNPGWorkspaceIssue is the subset of issuesapi.Issue the workspace renders. +type CNPGWorkspaceIssue struct { + ID string `json:"id"` + Severity issuesapi.Severity `json:"severity"` + Category issuesapi.Category `json:"category"` + Kind string `json:"kind"` + Group string `json:"group,omitempty"` + Namespace string `json:"namespace,omitempty"` + Name string `json:"name"` + Reason string `json:"reason"` + Message string `json:"message,omitempty"` + Cause string `json:"cause,omitempty"` + Action string `json:"action,omitempty"` + FirstSeen time.Time `json:"first_seen,omitzero"` +} + +// CNPGWorkspaceAuditFinding is one audit finding on a visible CNPG object. +type CNPGWorkspaceAuditFinding struct { + CheckID string `json:"checkId"` + Severity string `json:"severity"` + Kind string `json:"kind"` + Group string `json:"group,omitempty"` + Namespace string `json:"namespace"` + Name string `json:"name"` + Message string `json:"message"` +} + +// CNPGWorkspaceResponse is GET /api/cnpg/workspace. +type CNPGWorkspaceResponse struct { + Installed bool `json:"installed"` + Context string `json:"context"` + Namespaces []string `json:"namespaces"` + Coverage map[string]CNPGWorkspaceCoverage `json:"coverage"` + Objects map[string][]any `json:"objects"` + Issues []CNPGWorkspaceIssue `json:"issues"` + Audit []CNPGWorkspaceAuditFinding `json:"audit"` + BackupsOmitted int `json:"backupsOmitted"` +} + +// cnpgKindAccess is the resolved read scope for one kind. all means every +// namespace in the request's scope (or the cluster-scoped kind itself). +type cnpgKindAccess struct { + state string + all bool + namespaces map[string]bool +} + +func (a cnpgKindAccess) covers(namespace string) bool { + if a.state != cnpgCoverageFull && a.state != cnpgCoveragePartial { + return false + } + return a.all || a.namespaces[namespace] +} + +func newCNPGWorkspaceResponse(namespaces []string) CNPGWorkspaceResponse { + resp := CNPGWorkspaceResponse{ + Context: k8s.ActiveClusterContext(), + Namespaces: namespaces, + Coverage: map[string]CNPGWorkspaceCoverage{}, + Objects: map[string][]any{}, + Issues: []CNPGWorkspaceIssue{}, + Audit: []CNPGWorkspaceAuditFinding{}, + } + for _, k := range cnpgWorkspaceKinds { + resp.Coverage[k.key] = CNPGWorkspaceCoverage{State: cnpgCoverageNotInstalled} + resp.Objects[k.key] = []any{} + } + resp.Coverage[cnpgWorkspacePodsKey] = CNPGWorkspaceCoverage{State: cnpgCoverageNotInstalled} + resp.Objects[cnpgWorkspacePodsKey] = []any{} + return resp +} + +// handleCNPGWorkspace serves GET /api/cnpg/workspace: every CloudNativePG kind +// plus instance Pods, each authorized on its own. The generic resource list +// does not gate namespaced CRDs per kind, so it cannot tell "no access" from +// "none"; this endpoint states which one it is for every kind. +func (s *Server) handleCNPGWorkspace(w http.ResponseWriter, r *http.Request) { + if !s.requireConnected(w) { + return + } + cache := k8s.GetResourceCache() + if cache == nil { + s.writeError(w, http.StatusServiceUnavailable, "Resource cache not available") + return + } + + namespaces := s.parseNamespacesForUser(r) + resp := newCNPGWorkspaceResponse(namespaces) + + disc := k8s.GetResourceDiscovery() + if disc != nil { + for _, k := range cnpgWorkspaceKinds { + if _, ok := disc.GetGVRWithGroup(k.kind, k.group); ok { + resp.Installed = true + break + } + } + if !resp.Installed { + s.writeJSON(w, resp) + return + } + } + + access := map[string]cnpgKindAccess{} + items := map[string][]*unstructured.Unstructured{} + for _, k := range cnpgWorkspaceKinds { + if disc != nil { + if _, ok := disc.GetGVRWithGroup(k.kind, k.group); !ok { + access[k.key] = cnpgKindAccess{state: cnpgCoverageNotInstalled} + continue + } + } + acc, denied, list := s.cnpgWorkspaceReadKind(r, cache, k, namespaces) + if acc.state != cnpgCoverageNotInstalled { + resp.Installed = true + } + access[k.key] = acc + items[k.key] = list + resp.Coverage[k.key] = CNPGWorkspaceCoverage{State: acc.state, DeniedNamespaces: denied} + } + if !resp.Installed { + s.writeJSON(w, resp) + return + } + + for _, k := range cnpgWorkspaceKinds { + list := items[k.key] + if k.key == cnpgWorkspaceBackupsKey { + var omitted int + list, omitted = windowCNPGBackups(list, time.Now()) + resp.BackupsOmitted = omitted + } else { + sortCNPGObjects(list) + } + out := make([]any, 0, len(list)) + for _, u := range list { + out = append(out, u.Object) + } + resp.Objects[k.key] = out + } + + podAccess, podDenied, pods := s.cnpgWorkspaceReadPods(r, cache, namespaces) + access[cnpgWorkspacePodsKey] = podAccess + resp.Coverage[cnpgWorkspacePodsKey] = CNPGWorkspaceCoverage{State: podAccess.state, DeniedNamespaces: podDenied} + resp.Objects[cnpgWorkspacePodsKey] = pods + + resp.Issues = s.cnpgWorkspaceIssues(r, namespaces, access) + resp.Audit = cnpgWorkspaceAudit(items[cnpgWorkspaceClusterKey], items[cnpgWorkspaceSchedKey], access[cnpgWorkspaceSchedKey]) + + s.writeJSON(w, resp) +} + +// cnpgWorkspaceScope resolves where the caller may list one namespaced +// resource: nil allowed means the whole request scope. denied only ever names +// namespaces drawn from the caller's own scope (their view filter, or all +// namespaces for a caller who may see every namespace). +func (s *Server) cnpgWorkspaceScope(r *http.Request, namespaces []string, group, resource string) (allowed, denied []string, any bool) { + if noNamespaceAccess(namespaces) { + return []string{}, nil, false + } + if s.canRead(r, group, resource, "", "list") { + return namespaces, nil, true + } + candidates := namespaces + if candidates == nil { + candidates = allNamespaceNames() + } + if len(candidates) == 0 { + return []string{}, nil, false + } + allowed = s.filterNamespacesByCanRead(r, group, resource, "list", candidates) + for _, ns := range candidates { + if !slices.Contains(allowed, ns) { + denied = append(denied, ns) + } + } + sort.Strings(denied) + return allowed, denied, len(allowed) > 0 +} + +func accessFromScope(allowed, denied []string) cnpgKindAccess { + acc := cnpgKindAccess{state: cnpgCoverageFull, all: allowed == nil} + if len(denied) > 0 { + acc.state = cnpgCoveragePartial + } + if allowed != nil { + acc.namespaces = make(map[string]bool, len(allowed)) + for _, ns := range allowed { + acc.namespaces[ns] = true + } + } + return acc +} + +func (s *Server) cnpgWorkspaceReadKind(r *http.Request, cache *k8s.ResourceCache, k cnpgWorkspaceKind, namespaces []string) (cnpgKindAccess, []string, []*unstructured.Unstructured) { + var acc cnpgKindAccess + var denied, readNamespaces []string + if k.clusterScoped { + if !s.canRead(r, k.group, k.resource, "", "list") { + return cnpgKindAccess{state: cnpgCoverageDenied}, nil, nil + } + acc = cnpgKindAccess{state: cnpgCoverageFull, all: true} + } else { + allowed, d, ok := s.cnpgWorkspaceScope(r, namespaces, k.group, k.resource) + if !ok { + return cnpgKindAccess{state: cnpgCoverageDenied}, nil, nil + } + acc, denied, readNamespaces = accessFromScope(allowed, d), d, allowed + } + + list, err := readCNPGKind(r.Context(), cache, k, readNamespaces) + switch { + case err == nil: + return acc, denied, list + case errors.Is(err, k8s.ErrUnknownDynamicKind): + return cnpgKindAccess{state: cnpgCoverageNotInstalled}, nil, nil + case errors.Is(err, errDynamicNotSynced): + return cnpgKindAccess{state: cnpgCoverageSyncing}, nil, nil + default: + log.Printf("[cnpg] Failed to list %s.%s for workspace: %v", k.kind, k.group, err) + return cnpgKindAccess{state: cnpgCoverageError}, nil, nil + } +} + +func readCNPGKind(ctx context.Context, cache *k8s.ResourceCache, k cnpgWorkspaceKind, namespaces []string) ([]*unstructured.Unstructured, error) { + if namespaces == nil { + return filterCNPGGroup(listDynamicSynced(ctx, cache, k.kind, k.group, "")) + } + var out []*unstructured.Unstructured + for _, ns := range namespaces { + list, err := filterCNPGGroup(listDynamicSynced(ctx, cache, k.kind, k.group, ns)) + if err != nil { + return nil, err + } + out = append(out, list...) + } + return out, nil +} + +// filterCNPGGroup drops anything whose apiVersion is not a CNPG group, so a +// Velero Backup or a CAPI Cluster can never ride along on a kind-name match. +func filterCNPGGroup(items []*unstructured.Unstructured, err error) ([]*unstructured.Unstructured, error) { + if err != nil { + return nil, err + } + out := items[:0:0] + for _, u := range items { + if u == nil { + continue + } + if g := u.GroupVersionKind().Group; g != cnpgGroup && g != cnpgBarmanGroup { + continue + } + out = append(out, u) + } + return out, nil +} + +func sortCNPGObjects(items []*unstructured.Unstructured) { + sort.SliceStable(items, func(i, j int) bool { + if items[i].GetNamespace() != items[j].GetNamespace() { + return items[i].GetNamespace() < items[j].GetNamespace() + } + return items[i].GetName() < items[j].GetName() + }) +} + +func cnpgBackupTime(u *unstructured.Unstructured) time.Time { + for _, field := range []string{"stoppedAt", "startedAt"} { + if v, _, _ := unstructured.NestedString(u.Object, "status", field); v != "" { + if t, err := time.Parse(time.RFC3339, v); err == nil { + return t + } + } + } + return u.GetCreationTimestamp().Time +} + +// windowCNPGBackups keeps every in-flight Backup, settled ones from the last +// week, and each Cluster's newest completed Backup whatever its age. Sorted by +// namespace, newest first within it. +func windowCNPGBackups(items []*unstructured.Unstructured, now time.Time) ([]*unstructured.Unstructured, int) { + newestCompleted := map[string]*unstructured.Unstructured{} + for _, u := range items { + if phase, _, _ := unstructured.NestedString(u.Object, "status", "phase"); phase != "completed" { + continue + } + clusterName, _, _ := unstructured.NestedString(u.Object, "spec", "cluster", "name") + key := u.GetNamespace() + "\x00" + clusterName + if cur, ok := newestCompleted[key]; !ok || cnpgBackupTime(u).After(cnpgBackupTime(cur)) { + newestCompleted[key] = u + } + } + keepNewest := make(map[*unstructured.Unstructured]bool, len(newestCompleted)) + for _, u := range newestCompleted { + keepNewest[u] = true + } + + cutoff := now.Add(-cnpgBackupWindow) + kept := make([]*unstructured.Unstructured, 0, len(items)) + omitted := 0 + for _, u := range items { + phase, _, _ := unstructured.NestedString(u.Object, "status", "phase") + settled := phase == "completed" || phase == "failed" + if !settled || keepNewest[u] || !cnpgBackupTime(u).Before(cutoff) { + kept = append(kept, u) + continue + } + omitted++ + } + sort.SliceStable(kept, func(i, j int) bool { + if kept[i].GetNamespace() != kept[j].GetNamespace() { + return kept[i].GetNamespace() < kept[j].GetNamespace() + } + ti, tj := cnpgBackupTime(kept[i]), cnpgBackupTime(kept[j]) + if !ti.Equal(tj) { + return ti.After(tj) + } + return kept[i].GetName() < kept[j].GetName() + }) + return kept, omitted +} + +type cnpgWorkspacePodMeta struct { + Name string `json:"name"` + Namespace string `json:"namespace"` + UID types.UID `json:"uid"` + Labels map[string]string `json:"labels,omitempty"` + OwnerReferences []metav1.OwnerReference `json:"ownerReferences,omitempty"` + CreationTimestamp metav1.Time `json:"creationTimestamp"` +} + +type cnpgWorkspaceContainerStatus struct { + Name string `json:"name"` + Ready bool `json:"ready"` + RestartCount int32 `json:"restartCount"` + State corev1.ContainerState `json:"state"` +} + +type cnpgWorkspacePod struct { + APIVersion string `json:"apiVersion"` + Kind string `json:"kind"` + Metadata cnpgWorkspacePodMeta `json:"metadata"` + Spec struct { + NodeName string `json:"nodeName,omitempty"` + } `json:"spec"` + Status struct { + Phase corev1.PodPhase `json:"phase,omitempty"` + PodIP string `json:"podIP,omitempty"` + StartTime *metav1.Time `json:"startTime,omitempty"` + Conditions []corev1.PodCondition `json:"conditions,omitempty"` + ContainerStatuses []cnpgWorkspaceContainerStatus `json:"containerStatuses,omitempty"` + } `json:"status"` +} + +// isCNPGInstancePod requires the controller-set ownerReference as well as the +// label: a label alone is something any workload can carry. +func isCNPGInstancePod(p *corev1.Pod) bool { + clusterName := p.Labels["cnpg.io/cluster"] + if clusterName == "" { + return false + } + for _, ref := range p.OwnerReferences { + if ref.Kind != "Cluster" || ref.Name != clusterName { + continue + } + if gv, err := schema.ParseGroupVersion(ref.APIVersion); err == nil && gv.Group == cnpgGroup { + return true + } + } + return false +} + +func trimCNPGPod(p *corev1.Pod) cnpgWorkspacePod { + out := cnpgWorkspacePod{APIVersion: "v1", Kind: "Pod"} + out.Metadata = cnpgWorkspacePodMeta{ + Name: p.Name, + Namespace: p.Namespace, + UID: p.UID, + Labels: p.Labels, + OwnerReferences: p.OwnerReferences, + CreationTimestamp: p.CreationTimestamp, + } + out.Spec.NodeName = p.Spec.NodeName + out.Status.Phase = p.Status.Phase + out.Status.PodIP = p.Status.PodIP + out.Status.StartTime = p.Status.StartTime + out.Status.Conditions = p.Status.Conditions + for _, cs := range p.Status.ContainerStatuses { + out.Status.ContainerStatuses = append(out.Status.ContainerStatuses, cnpgWorkspaceContainerStatus{ + Name: cs.Name, Ready: cs.Ready, RestartCount: cs.RestartCount, State: cs.State, + }) + } + return out +} + +func (s *Server) cnpgWorkspaceReadPods(r *http.Request, cache *k8s.ResourceCache, namespaces []string) (cnpgKindAccess, []string, []any) { + out := []any{} + allowed, denied, ok := s.cnpgWorkspaceScope(r, namespaces, "", "pods") + if !ok { + return cnpgKindAccess{state: cnpgCoverageDenied}, nil, out + } + if cache.Pods() == nil { + log.Printf("[cnpg] Pod cache unavailable for workspace") + return cnpgKindAccess{state: cnpgCoverageError}, nil, out + } + // The typed Pod informer may itself be namespace-scoped when Radar's own + // identity cannot list Pods cluster-wide; what it does not hold is unread. + within := capacityNamespacesWithinCache(cache, "pods", allowed) + if within.unavailable { + log.Printf("[cnpg] Pod cache does not cover the workspace scope") + return cnpgKindAccess{state: cnpgCoverageError}, nil, out + } + if allowed != nil { + for _, ns := range allowed { + if !slices.Contains(within.namespaces, ns) { + denied = append(denied, ns) + } + } + sort.Strings(denied) + } + acc := accessFromScope(within.namespaces, denied) + if within.partial { + acc.state = cnpgCoveragePartial + } + + pods := listPodsScoped(cache.Pods(), within.namespaces) + sort.Slice(pods, func(i, j int) bool { + if pods[i].Namespace != pods[j].Namespace { + return pods[i].Namespace < pods[j].Namespace + } + return pods[i].Name < pods[j].Name + }) + for _, p := range pods { + if p != nil && isCNPGInstancePod(p) { + out = append(out, trimCNPGPod(p)) + } + } + return acc, denied, out +} + +var cnpgWorkspaceKeyByGroupKind = func() map[string]string { + m := make(map[string]string, len(cnpgWorkspaceKinds)) + for _, k := range cnpgWorkspaceKinds { + m[k.group+"/"+k.kind] = k.key + } + return m +}() + +// cnpgWorkspaceIssues runs the same composition /api/issues serves, then keeps +// only CNPG subjects on a kind and namespace this response had coverage for — +// an issue on an object the caller could not list would disclose it. +func (s *Server) cnpgWorkspaceIssues(r *http.Request, namespaces []string, access map[string]cnpgKindAccess) []CNPGWorkspaceIssue { + out := []CNPGWorkspaceIssue{} + if noNamespaceAccess(namespaces) { + return out + } + provider := issues.NewCacheProvider() + if provider == nil { + return out + } + composed, _ := issues.ComposeWithStats(provider, issues.Filters{ + Namespaces: namespaces, + Limit: issues.NoLimit, + Grouped: true, + CanReadClusterScoped: s.issueClusterScopedAccess(r), + CanReadRelated: s.issueRelatedResourceAccess(r), + }) + for _, iss := range composed { + if iss.Group != cnpgGroup && iss.Group != cnpgBarmanGroup { + continue + } + key, ok := cnpgWorkspaceKeyByGroupKind[iss.Group+"/"+iss.Kind] + if !ok || !access[key].covers(iss.Namespace) { + continue + } + out = append(out, CNPGWorkspaceIssue{ + ID: iss.ID, + Severity: iss.Severity, + Category: iss.Category, + Kind: iss.Kind, + Group: iss.Group, + Namespace: iss.Namespace, + Name: iss.Name, + Reason: iss.Reason, + Message: iss.Message, + Cause: iss.Cause, + Action: iss.Action, + FirstSeen: iss.FirstSeen, + }) + } + return out +} + +// cnpgWorkspaceAudit reports the declarative-backup posture finding only for +// Clusters whose namespace had its ScheduledBackups read: without that list, +// "no schedule targets this cluster" is an absence nobody established. +func cnpgWorkspaceAudit(clusters, scheduled []*unstructured.Unstructured, schedAccess cnpgKindAccess) []CNPGWorkspaceAuditFinding { + out := []CNPGWorkspaceAuditFinding{} + var subjects []*unstructured.Unstructured + for _, c := range clusters { + if schedAccess.covers(c.GetNamespace()) { + subjects = append(subjects, c) + } + } + if len(subjects) == 0 { + return out + } + results := bp.RunChecks(&bp.CheckInput{ + CNPGClusters: subjects, + CNPGScheduledBackups: scheduled, + CNPGScheduledBackupsAuthoritative: true, + }) + results = applyAuditSettings(results, getAuditConfig()) + if results == nil { + return out + } + for _, f := range results.Findings { + if f.CheckID != cnpgNoDeclarativeBackupCheckID { + continue + } + out = append(out, CNPGWorkspaceAuditFinding{ + CheckID: f.CheckID, + Severity: f.Severity, + Kind: f.Kind, + Group: f.Group, + Namespace: f.Namespace, + Name: f.Name, + Message: f.Message, + }) + } + sort.SliceStable(out, func(i, j int) bool { + if out[i].Namespace != out[j].Namespace { + return out[i].Namespace < out[j].Namespace + } + return out[i].Name < out[j].Name + }) + return out +} diff --git a/internal/server/cnpg_workspace_test.go b/internal/server/cnpg_workspace_test.go new file mode 100644 index 000000000..ee2d4edf3 --- /dev/null +++ b/internal/server/cnpg_workspace_test.go @@ -0,0 +1,464 @@ +package server + +import ( + "context" + "encoding/json" + "net/http" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/runtime/schema" + dynamicfake "k8s.io/client-go/dynamic/fake" + + "github.com/skyhook-io/radar/internal/auth" + "github.com/skyhook-io/radar/internal/k8s" +) + +func cnpgTestResource(group, kind, resource string, namespaced bool) k8s.APIResource { + return k8s.APIResource{Group: group, Version: "v1", Kind: kind, Name: resource, Namespaced: namespaced, IsCRD: true, Verbs: []string{"get", "list", "watch"}} +} + +var cnpgWorkspaceTestKinds = func() []k8s.APIResource { + var out []k8s.APIResource + for _, k := range cnpgWorkspaceKinds { + out = append(out, cnpgTestResource(k.group, k.kind, k.resource, !k.clusterScoped)) + } + out = append(out, + cnpgTestResource(veleroGroup, "Backup", "backups", true), + k8s.APIResource{Group: "cluster.x-k8s.io", Version: "v1beta1", Kind: "Cluster", Name: "clusters", Namespaced: true, IsCRD: true, Verbs: []string{"get", "list", "watch"}}, + ) + return out +}() + +func seedCNPGWorkspace(t *testing.T, kinds []k8s.APIResource, objs ...runtime.Object) { + t.Helper() + listKinds := map[schema.GroupVersionResource]string{} + for _, k := range kinds { + listKinds[schema.GroupVersionResource{Group: k.Group, Version: k.Version, Resource: k.Name}] = k.Kind + "List" + } + dyn := dynamicfake.NewSimpleDynamicClientWithCustomListKinds(runtime.NewScheme(), listKinds, objs...) + if err := k8s.InitTestDynamicResourceCache(dyn, kinds); err != nil { + t.Fatalf("seed cnpg: %v", err) + } + t.Cleanup(k8s.ResetTestDynamicState) +} + +func cnpgObj(apiVersion, kind, ns, name string, spec, status map[string]any) *unstructured.Unstructured { + meta := map[string]any{"name": name, "creationTimestamp": time.Now().Add(-time.Hour).UTC().Format(time.RFC3339)} + if ns != "" { + meta["namespace"] = ns + } + obj := map[string]any{"apiVersion": apiVersion, "kind": kind, "metadata": meta} + if spec != nil { + obj["spec"] = spec + } + if status != nil { + obj["status"] = status + } + return &unstructured.Unstructured{Object: obj} +} + +func cnpgBackup(ns, name, cluster, phase string, stoppedAt time.Time) *unstructured.Unstructured { + status := map[string]any{"phase": phase} + if !stoppedAt.IsZero() { + status["stoppedAt"] = stoppedAt.UTC().Format(time.RFC3339) + } + return cnpgObj("postgresql.cnpg.io/v1", "Backup", ns, name, map[string]any{"cluster": map[string]any{"name": cluster}}, status) +} + +func decodeWorkspace(t *testing.T, resp *http.Response) CNPGWorkspaceResponse { + t.Helper() + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK { + t.Fatalf("status = %d, want 200", resp.StatusCode) + } + var out CNPGWorkspaceResponse + if err := json.NewDecoder(resp.Body).Decode(&out); err != nil { + t.Fatalf("decode: %v", err) + } + return out +} + +func getWorkspaceNoAuth(t *testing.T, query string) CNPGWorkspaceResponse { + t.Helper() + resp, err := http.Get(testServer.URL + "/api/cnpg/workspace" + query) + if err != nil { + t.Fatalf("GET: %v", err) + } + return decodeWorkspace(t, resp) +} + +func objectNames(objs []any) []string { + var out []string + for _, o := range objs { + m, _ := o.(map[string]any) + meta, _ := m["metadata"].(map[string]any) + name, _ := meta["name"].(string) + out = append(out, name) + } + return out +} + +func containsName(objs []any, name string) bool { + for _, n := range objectNames(objs) { + if n == name { + return true + } + } + return false +} + +func assertEveryKey(t *testing.T, got CNPGWorkspaceResponse) { + t.Helper() + keys := []string{cnpgWorkspacePodsKey} + for _, k := range cnpgWorkspaceKinds { + keys = append(keys, k.key) + } + for _, k := range keys { + if _, ok := got.Coverage[k]; !ok { + t.Errorf("coverage missing key %q", k) + } + if objs, ok := got.Objects[k]; !ok || objs == nil { + t.Errorf("objects missing key %q (or null)", k) + } + } +} + +func TestCNPGWorkspace_NotInstalled(t *testing.T) { + seedCNPGWorkspace(t, []k8s.APIResource{cnpgTestResource(veleroGroup, "Backup", "backups", true)}) + got := getWorkspaceNoAuth(t, "") + if got.Installed { + t.Error("installed = true on a cluster without CloudNativePG") + } + assertEveryKey(t, got) + for k, c := range got.Coverage { + if c.State != cnpgCoverageNotInstalled { + t.Errorf("coverage[%s] = %q, want notInstalled", k, c.State) + } + } +} + +func seedCNPGPods(t *testing.T, pods ...*corev1.Pod) { + t.Helper() + ctx := context.Background() + for _, p := range pods { + if _, err := testFakeClient.CoreV1().Pods(p.Namespace).Create(ctx, p, metav1.CreateOptions{}); err != nil { + t.Fatalf("create pod %s: %v", p.Name, err) + } + t.Cleanup(func() { + _ = testFakeClient.CoreV1().Pods(p.Namespace).Delete(context.Background(), p.Name, metav1.DeleteOptions{}) + }) + } + lister := k8s.GetResourceCache().Pods() + deadline := time.Now().Add(5 * time.Second) + for { + seen := 0 + for _, p := range pods { + if _, err := lister.Pods(p.Namespace).Get(p.Name); err == nil { + seen++ + } + } + if seen == len(pods) { + return + } + if time.Now().After(deadline) { + t.Fatalf("pods did not reach the cache") + } + time.Sleep(20 * time.Millisecond) + } +} + +func cnpgPod(ns, name, clusterLabel string, owners ...metav1.OwnerReference) *corev1.Pod { + return &corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{ + Name: name, Namespace: ns, + Labels: map[string]string{"cnpg.io/cluster": clusterLabel, "cnpg.io/instanceRole": "primary"}, + OwnerReferences: owners, + }, + Spec: corev1.PodSpec{NodeName: "node-1", Containers: []corev1.Container{{Name: "postgres", Image: "pg:17"}}}, + Status: corev1.PodStatus{ + Phase: corev1.PodRunning, + PodIP: "10.0.0.5", + ContainerStatuses: []corev1.ContainerStatus{{ + Name: "postgres", Ready: true, RestartCount: 2, Image: "pg:17", + State: corev1.ContainerState{Running: &corev1.ContainerStateRunning{}}, + }}, + }, + } +} + +func TestCNPGWorkspace_AuthDisabledReturnsEverythingAndOnlyOwnedInstancePods(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgws", "pg-orders", map[string]any{"instances": int64(1)}, nil), + cnpgObj("postgresql.cnpg.io/v1", "Pooler", "pgws", "pg-orders-rw", map[string]any{"cluster": map[string]any{"name": "pg-orders"}}, nil), + cnpgObj("postgresql.cnpg.io/v1", "ClusterImageCatalog", "", "pg-fleet", nil, nil), + cnpgObj("barmancloud.cnpg.io/v1", "ObjectStore", "pgws", "store", nil, nil), + ) + owner := metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "c-uid", Controller: boolPtr(true)} + seedCNPGPods(t, + cnpgPod("pgws", "pg-orders-1", "pg-orders", owner), + cnpgPod("pgws", "impostor-1", "pg-orders"), + cnpgPod("pgws", "capi-owned-1", "pg-orders", metav1.OwnerReference{APIVersion: "cluster.x-k8s.io/v1beta1", Kind: "Cluster", Name: "pg-orders", UID: "x"}), + cnpgPod("pgws", "other-owner-1", "pg-orders", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-billing", UID: "y"}), + ) + + got := getWorkspaceNoAuth(t, "") + if !got.Installed { + t.Fatal("installed = false with CNPG kinds discovered") + } + assertEveryKey(t, got) + for k, c := range got.Coverage { + if c.State != cnpgCoverageFull { + t.Errorf("coverage[%s] = %+v, want full with auth disabled", k, c) + } + } + if got.Namespaces != nil { + t.Errorf("namespaces = %v, want null for an unfiltered view", got.Namespaces) + } + for key, name := range map[string]string{"clusters": "pg-orders", "poolers": "pg-orders-rw", "clusterImageCatalogs": "pg-fleet", "objectStores": "store"} { + if !containsName(got.Objects[key], name) { + t.Errorf("objects[%s] = %v, want %s", key, objectNames(got.Objects[key]), name) + } + } + pods := objectNames(got.Objects["pods"]) + if len(pods) != 1 || pods[0] != "pg-orders-1" { + t.Fatalf("pods = %v, want only the CNPG-owned instance pod", pods) + } + pod := got.Objects["pods"][0].(map[string]any) + if _, ok := pod["spec"].(map[string]any)["containers"]; ok { + t.Error("pod spec was not trimmed") + } + cs := pod["status"].(map[string]any)["containerStatuses"].([]any)[0].(map[string]any) + if cs["restartCount"].(float64) != 2 || cs["ready"] != true { + t.Errorf("container status = %v", cs) + } + if _, ok := cs["image"]; ok { + t.Error("container status carried fields beyond the trimmed set") + } +} + +func TestCNPGWorkspace_CollidingKindsNeverLeak(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgws", "pg-orders", nil, nil), + cnpgBackup("pgws", "pg-orders-b1", "pg-orders", "running", time.Time{}), + cnpgObj("velero.io/v1", "Backup", "pgws", "velero-nightly", nil, map[string]any{"phase": "Completed"}), + cnpgObj("cluster.x-k8s.io/v1beta1", "Cluster", "pgws", "capi-workload", nil, nil), + ) + got := getWorkspaceNoAuth(t, "") + if containsName(got.Objects["backups"], "velero-nightly") { + t.Error("a Velero Backup was returned as a CNPG Backup") + } + if containsName(got.Objects["clusters"], "capi-workload") { + t.Error("a CAPI Cluster was returned as a CNPG Cluster") + } + if !containsName(got.Objects["backups"], "pg-orders-b1") || !containsName(got.Objects["clusters"], "pg-orders") { + t.Errorf("CNPG objects missing: backups=%v clusters=%v", objectNames(got.Objects["backups"]), objectNames(got.Objects["clusters"])) + } +} + +func TestCNPGWorkspace_DeniedKindAndItsIssuesAreWithheld(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pg", "pg-orders", nil, nil), + cnpgBackup("pg", "pg-orders-broken", "pg-orders", "failed", time.Now().Add(-time.Hour)), + ) + // Warm the Backup informer so the issues engine can see the failed Backup + // whichever caller asks; otherwise a denied answer would pass vacuously. + if _, err := listDynamicSynced(context.Background(), k8s.GetResourceCache(), "Backup", cnpgGroup, ""); err != nil { + t.Fatalf("warm backups: %v", err) + } + + env := newAuthTestServer(t) + for _, u := range []struct { + name string + backupsListed bool + }{{"reader", true}, {"no-backups", false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pg"}} + allow(perms, cnpgGroup, "clusters", "", true) + allow(perms, cnpgGroup, "backups", "", u.backupsListed) + allow(perms, cnpgGroup, "backups", "pg", u.backupsListed) + env.srv.permCache.Set(u.name, nil, perms) + } + + hasBackupIssue := func(got CNPGWorkspaceResponse) bool { + for _, iss := range got.Issues { + if iss.Kind == "Backup" && iss.Name == "pg-orders-broken" { + return true + } + } + return false + } + + control := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "reader", "")) + if control.Coverage["backups"].State != cnpgCoverageFull || !containsName(control.Objects["backups"], "pg-orders-broken") { + t.Fatalf("control: backups coverage=%+v objects=%v", control.Coverage["backups"], objectNames(control.Objects["backups"])) + } + if !hasBackupIssue(control) { + t.Fatalf("control: the failed Backup raised no issue, so the denied case would prove nothing: %+v", control.Issues) + } + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "no-backups", "")) + if got.Coverage["backups"].State != cnpgCoverageDenied { + t.Errorf("backups coverage = %+v, want denied", got.Coverage["backups"]) + } + if len(got.Objects["backups"]) != 0 { + t.Errorf("backups = %v, want [] when denied", objectNames(got.Objects["backups"])) + } + if hasBackupIssue(got) { + t.Error("an issue on a Backup the caller cannot list was returned") + } + if got.Coverage["clusters"].State != cnpgCoverageFull || !containsName(got.Objects["clusters"], "pg-orders") { + t.Errorf("clusters coverage=%+v objects=%v, want full", got.Coverage["clusters"], objectNames(got.Objects["clusters"])) + } +} + +func TestCNPGWorkspace_PartialNamespaceCoverage(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "a", "pg-a", nil, nil), + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "b", "pg-b", nil, nil), + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "c", "pg-c", nil, nil), + ) + env := newAuthTestServer(t) + perms := &auth.UserPermissions{AllowedNamespaces: []string{"a", "b"}} + allow(perms, cnpgGroup, "clusters", "", false) + allow(perms, cnpgGroup, "clusters", "a", true) + allow(perms, cnpgGroup, "clusters", "b", false) + env.srv.permCache.Set("scoped", nil, perms) + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "scoped", "")) + cov := got.Coverage["clusters"] + if cov.State != cnpgCoveragePartial || len(cov.DeniedNamespaces) != 1 || cov.DeniedNamespaces[0] != "b" { + t.Errorf("clusters coverage = %+v, want partial denied [b]", cov) + } + names := objectNames(got.Objects["clusters"]) + if len(names) != 1 || names[0] != "pg-a" { + t.Errorf("clusters = %v, want only pg-a", names) + } + + // A view filter narrows the scope; the denied list never grows past it. + filtered := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace?namespaces=a", "scoped", "")) + if filtered.Coverage["clusters"].State != cnpgCoverageFull { + t.Errorf("filtered to a: coverage = %+v, want full", filtered.Coverage["clusters"]) + } + if len(filtered.Namespaces) != 1 || filtered.Namespaces[0] != "a" { + t.Errorf("namespaces = %v, want [a]", filtered.Namespaces) + } +} + +func TestCNPGWorkspace_ClusterImageCatalogNeedsClusterScopeGrant(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "ClusterImageCatalog", "", "pg-fleet", nil, nil), + ) + env := newAuthTestServer(t) + nsOnly := &auth.UserPermissions{AllowedNamespaces: []string{"pg"}} + nsOnly.SetCanI("list", cnpgGroup, "clusterimagecatalogs", "pg", true) + allow(nsOnly, cnpgGroup, "clusterimagecatalogs", "", false) + env.srv.permCache.Set("ns-only", nil, nsOnly) + + clusterWide := &auth.UserPermissions{AllowedNamespaces: []string{"pg"}} + allow(clusterWide, cnpgGroup, "clusterimagecatalogs", "", true) + env.srv.permCache.Set("cluster-wide", nil, clusterWide) + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "ns-only", "")) + if got.Coverage["clusterImageCatalogs"].State != cnpgCoverageDenied || len(got.Objects["clusterImageCatalogs"]) != 0 { + t.Errorf("namespace-level grant exposed ClusterImageCatalogs: %+v %v", got.Coverage["clusterImageCatalogs"], objectNames(got.Objects["clusterImageCatalogs"])) + } + + got = decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace?namespaces=pg", "cluster-wide", "")) + if got.Coverage["clusterImageCatalogs"].State != cnpgCoverageFull || !containsName(got.Objects["clusterImageCatalogs"], "pg-fleet") { + t.Errorf("cluster-scope grant: %+v %v, want full with pg-fleet regardless of the view filter", got.Coverage["clusterImageCatalogs"], objectNames(got.Objects["clusterImageCatalogs"])) + } +} + +func TestWindowCNPGBackups(t *testing.T) { + now := time.Date(2026, 9, 29, 12, 0, 0, 0, time.UTC) + day := 24 * time.Hour + items := []*unstructured.Unstructured{ + cnpgBackup("pg", "running-old", "orders", "running", time.Time{}), + cnpgBackup("pg", "orders-recent", "orders", "completed", now.Add(-2*day)), + cnpgBackup("pg", "orders-old", "orders", "completed", now.Add(-20*day)), + cnpgBackup("pg", "orders-failed-old", "orders", "failed", now.Add(-30*day)), + cnpgBackup("pg", "orders-failed-new", "orders", "failed", now.Add(-1*day)), + cnpgBackup("pg", "billing-only-old", "billing", "completed", now.Add(-40*day)), + cnpgBackup("pg", "billing-older", "billing", "completed", now.Add(-50*day)), + cnpgBackup("aa", "other-ns", "orders", "completed", now.Add(-60*day)), + } + // A running Backup older than the window stays: in flight is never settled. + items[0].Object["metadata"].(map[string]any)["creationTimestamp"] = now.Add(-90 * day).Format(time.RFC3339) + + kept, omitted := windowCNPGBackups(items, now) + var names []string + for _, u := range kept { + names = append(names, u.GetName()) + } + want := []string{"other-ns", "orders-failed-new", "orders-recent", "billing-only-old", "running-old"} + if len(names) != len(want) { + t.Fatalf("kept = %v, want %v", names, want) + } + for i := range want { + if names[i] != want[i] { + t.Fatalf("kept = %v, want %v (namespace, then newest first)", names, want) + } + } + if omitted != 3 { + t.Errorf("omitted = %d, want 3 (orders-old, orders-failed-old, billing-older)", omitted) + } +} + +func TestCNPGWorkspace_BackupsOmittedIsReported(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgBackup("pgws", "new", "orders", "completed", time.Now().Add(-time.Hour)), + cnpgBackup("pgws", "ancient", "orders", "completed", time.Now().Add(-30*24*time.Hour)), + ) + got := getWorkspaceNoAuth(t, "") + if got.BackupsOmitted != 1 || containsName(got.Objects["backups"], "ancient") { + t.Errorf("backupsOmitted=%d backups=%v, want 1 and ancient omitted", got.BackupsOmitted, objectNames(got.Objects["backups"])) + } +} + +func TestCNPGWorkspace_AuditNeedsScheduledBackupEvidence(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pg", "unscheduled", nil, nil), + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pg", "scheduled", nil, nil), + cnpgObj("postgresql.cnpg.io/v1", "ScheduledBackup", "pg", "nightly", map[string]any{"cluster": map[string]any{"name": "scheduled"}}, nil), + ) + env := newAuthTestServer(t) + for _, u := range []struct { + name string + sched bool + }{{"sees-schedules", true}, {"no-schedules", false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pg"}} + allow(perms, cnpgGroup, "clusters", "", true) + allow(perms, cnpgGroup, "scheduledbackups", "", u.sched) + allow(perms, cnpgGroup, "scheduledbackups", "pg", u.sched) + env.srv.permCache.Set(u.name, nil, perms) + } + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "sees-schedules", "")) + if len(got.Audit) != 1 || got.Audit[0].Name != "unscheduled" || got.Audit[0].CheckID != cnpgNoDeclarativeBackupCheckID { + t.Errorf("audit = %+v, want one %s finding on unscheduled", got.Audit, cnpgNoDeclarativeBackupCheckID) + } + + got = decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "no-schedules", "")) + if got.Coverage["scheduledBackups"].State != cnpgCoverageDenied { + t.Errorf("scheduledBackups coverage = %+v, want denied", got.Coverage["scheduledBackups"]) + } + if len(got.Audit) != 0 { + t.Errorf("audit = %+v, want none — without the ScheduledBackup list the absence is unestablished", got.Audit) + } +} + +func TestIsCNPGInstancePod(t *testing.T) { + owned := cnpgPod("pg", "x-1", "x", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "x"}) + if !isCNPGInstancePod(owned) { + t.Error("owned instance pod rejected") + } + unlabeled := owned.DeepCopy() + unlabeled.Labels = nil + if isCNPGInstancePod(unlabeled) { + t.Error("pod without the cluster label accepted") + } +} diff --git a/internal/server/server.go b/internal/server/server.go index 4336bfae8..3fde11d61 100644 --- a/internal/server/server.go +++ b/internal/server/server.go @@ -591,6 +591,7 @@ func (s *Server) setupAppRoutes(r chi.Router) { r.Get("/rbac/role/{kind}/{namespace}/{name}", s.handleRBACRole) r.Get("/rbac/namespace/{namespace}", s.handleRBACNamespace) r.Get("/rbac/whoami", s.handleRBACWhoami) + r.Get("/cnpg/workspace", s.handleCNPGWorkspace) r.Get("/cnpg/imagecatalogs/{namespace}/{name}/clusters", s.handleCNPGCatalogUsers) r.Get("/cnpg/clusterimagecatalogs/{name}/clusters", s.handleCNPGCatalogUsers) r.Get("/velero/backupstoragelocations/{namespace}/{name}/backups", s.handleVeleroStoredBackups) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx new file mode 100644 index 000000000..1bf30a1f5 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx @@ -0,0 +1,220 @@ +import type { ReactNode } from 'react' +import { clsx } from 'clsx' +import { Badge } from '../ui/Badge' +import { Tooltip } from '../ui/Tooltip' +import { CNPG_BARMAN_OBJECTSTORE_GROUP, CNPG_GROUP } from '../resources/resource-utils-cnpg' +import type { CNPGFleetRow, CNPGInstance } from './workspace' +import { + FactGrid, + FactRow, + FactSource, + FactValue, + ProblemCallout, + RefLink, + SummaryHeading, + ToneDot, + toneTextClass, + CNPG_PRIMARY_BUTTON, + CNPG_SECONDARY_BUTTON, + type CNPGNavigate, +} from './primitives' + +export interface CNPGSummaryAction { + label: string + onClick: () => void + primary?: boolean +} + +function InstancePill({ pod, namespace, onNavigate }: { pod: CNPGInstance; namespace: string; onNavigate?: CNPGNavigate }) { + const tone = pod.ready === true ? 'healthy' : pod.ready === false ? 'unhealthy' : 'unknown' + const role = pod.role === 'primary' ? 'Primary' : pod.role === 'replica' ? 'Replica' : 'Role unknown' + const readiness = pod.ready === true ? 'Ready' : pod.ready === false ? 'Not ready' : 'Readiness unknown' + return ( + + + + ) +} + +export function CNPGClusterSummary({ + row, + onNavigate, + actions, + problemsLink, + extra, +}: { + row: CNPGFleetRow + onNavigate?: CNPGNavigate + actions?: CNPGSummaryAction[] + /** Link to the complete list of this cluster's findings, shown when more than one exists. */ + problemsLink?: (count: number) => ReactNode + extra?: ReactNode +}) { + const top = row.problems[0] + const rest = row.problems.length - 1 + const p = row.protection + const ns = row.namespace + const radarFindings = row.problems.some((x) => x.severity !== 'posture') + + return ( +
+ {top && ( + 0 ? problemsLink?.(row.problems.length) ?? +{rest} more : null} + /> + )} + + {actions && actions.length > 0 && ( +
+ {actions.map((a) => ( + + ))} +
+ )} + + State + + + + + {row.controllerStatus.text} + + + {radarFindings ? 'reported by CNPG · Radar findings above are separate' : 'reported by CNPG'} + + + + +
+ + {row.instances.ready ?? '–'}/{row.instances.desired ?? '–'} ready + {row.cluster?.status?.currentPrimary && ( + · primary {row.cluster.status.currentPrimary} + )} + + {row.pods.length > 0 && ( +
+ {row.pods.map((pod) => ( + + ))} +
+ )} +
+
+ + + + + {row.replicaCluster && ( + + Follows {row.replicaCluster.source ? {row.replicaCluster.source} : 'an external primary'} + + )} + + {row.pgVersion ?? 'Unknown'} + {row.catalog && ( + + {' · '} + + {row.catalog.name} + + + )} + + + + + + {row.poolers.length === 0 ? ( + None + ) : ( + + {row.poolers.map((name) => ( + + ))} + + )} + + {row.gitops && ( + + {row.gitops.tool === 'argocd' ? 'Argo CD' : 'Flux'} {row.gitops.name} + + )} +
+ + Protection + + + + {p.schedule.names.length > 0 && ( +
+ {p.schedule.names.map((name) => ( + + ))} +
+ )} +
+ + {p.destination.objectStore ? ( + + ObjectStore {p.destination.objectStore} + + ) : ( + + )} + + + + + + + + + + {p.recoveryWindow.from ? ( +
+ + {new Date(p.recoveryWindow.from).toUTCString().replace(' GMT', ' UTC')} → {p.recoveryWindow.to ? new Date(p.recoveryWindow.to).toUTCString().replace(' GMT', ' UTC') : 'unknown'} + + + {p.recoveryWindow.tone === 'degraded' && ( +
A backup failed after the last success, so this window is not advancing.
+ )} +
+ ) : ( + + )} +
+ + {p.restoreValidation.restoredInto ? ( + + {p.restoreValidation.text} + + ) : ( + + )} + + +
+ {extra} +
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/index.ts b/packages/k8s-ui/src/components/cnpg/index.ts new file mode 100644 index 000000000..adbdba2df --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/index.ts @@ -0,0 +1,3 @@ +export * from './workspace' +export * from './primitives' +export * from './CNPGClusterSummary' diff --git a/packages/k8s-ui/src/components/cnpg/primitives.tsx b/packages/k8s-ui/src/components/cnpg/primitives.tsx new file mode 100644 index 000000000..711e5680c --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/primitives.tsx @@ -0,0 +1,131 @@ +import type { ReactNode } from 'react' +import { clsx } from 'clsx' +import type { HealthLevel } from '../resources/resource-utils' +import { formatAge } from '../resources/resource-utils' +import { StatusDot } from '../ui/status-tone' +import { Tooltip } from '../ui/Tooltip' +import { AlertBanner } from '../ui/drawer-components' +import { TONE_TEXT_CLASS } from '../ui/severity-tone' +import type { CNPGFact, CNPGProblem } from './workspace' + +export interface CNPGRef { + kind: string + group?: string + namespace: string + name: string +} + +export type CNPGNavigate = (ref: CNPGRef) => void + +const TONE_TEXT: Record = { + healthy: 'text-theme-text-primary', + degraded: TONE_TEXT_CLASS.amber, + alert: TONE_TEXT_CLASS.orange, + unhealthy: TONE_TEXT_CLASS.red, + unknown: 'text-theme-text-tertiary', + neutral: 'text-theme-text-secondary', +} + +export const CNPG_PRIMARY_BUTTON = 'btn-brand inline-flex items-center gap-1.5 px-3 py-1.5 text-sm font-medium' +export const CNPG_SECONDARY_BUTTON = + 'inline-flex items-center gap-1.5 rounded-lg border border-theme-border bg-theme-surface px-3 py-1.5 text-sm text-theme-text-primary transition-colors hover:bg-theme-hover' + +export function toneTextClass(tone: HealthLevel): string { + return TONE_TEXT[tone] +} + +export function FactValue({ fact, className }: { fact: CNPGFact; className?: string }) { + const age = fact.at ? formatAge(fact.at) : null + const body = ( + + {fact.text} + {age && {fact.text ? ' · ' : ''}{age} ago} + + ) + if (!fact.source && !fact.at) return body + return ( + + {body} + + ) +} + +export function FactSource({ fact }: { fact: CNPGFact }) { + if (!fact.source) return null + return
{fact.source}
+} + +export function FactGrid({ children }: { children: ReactNode }) { + return
{children}
+} + +export function FactRow({ label, children }: { label: ReactNode; children: ReactNode }) { + return ( + <> +
{label}
+
{children}
+ + ) +} + +export function SummaryHeading({ children, hint }: { children: ReactNode; hint?: ReactNode }) { + return ( +
+

{children}

+ {hint && {hint}} +
+ ) +} + +export function RefLink({ refTo, onNavigate, children, mono }: { refTo: CNPGRef; onNavigate?: CNPGNavigate; children?: ReactNode; mono?: boolean }) { + const label = children ?? refTo.name + if (!onNavigate) return {label} + return ( + + ) +} + +const PROBLEM_VARIANT: Record = { + critical: 'error', + warning: 'warning', + posture: 'info', +} + +export function ProblemCallout({ + problem, + more, + onNavigate, + action, +}: { + problem: CNPGProblem + more?: ReactNode + onNavigate?: CNPGNavigate + action?: ReactNode +}) { + const aboutChild = problem.subject.kind !== 'Cluster' + return ( + +
+ {aboutChild && ( + + {problem.subject.kind}{' '} + + + )} + {problem.source === 'audit' ? 'Radar check' : 'Radar issue'} + {action} + {more} +
+
+ ) +} + +export function ToneDot({ tone }: { tone: HealthLevel }) { + return +} diff --git a/packages/k8s-ui/src/components/cnpg/workspace.test.ts b/packages/k8s-ui/src/components/cnpg/workspace.test.ts new file mode 100644 index 000000000..5004ec3bb --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/workspace.test.ts @@ -0,0 +1,179 @@ +import { describe, it, expect } from 'vitest' +import { buildCNPGFleet, type CNPGWorkspaceResponse, type CNPGWorkspaceKey, CNPG_WORKSPACE_KEYS } from './workspace' + +const G = 'postgresql.cnpg.io/v1' + +function cluster(name: string, ns: string, extra: any = {}): any { + return { + apiVersion: G, + kind: 'Cluster', + metadata: { name, namespace: ns, ...(extra.metadata ?? {}) }, + spec: { instances: 3, ...(extra.spec ?? {}) }, + status: { + phase: 'Cluster in healthy state', + readyInstances: 3, + currentPrimary: `${name}-1`, + ...(extra.status ?? {}), + }, + } +} + +function pod(name: string, ns: string, clusterName: string, role: string, ready = true): any { + return { + apiVersion: 'v1', + kind: 'Pod', + metadata: { name, namespace: ns, labels: { 'cnpg.io/cluster': clusterName, 'cnpg.io/instanceRole': role } }, + status: { conditions: [{ type: 'Ready', status: ready ? 'True' : 'False' }] }, + } +} + +function resp(objects: Partial>, over: Partial = {}): CNPGWorkspaceResponse { + const coverage: CNPGWorkspaceResponse['coverage'] = {} + for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } + return { + installed: true, + context: 'test', + namespaces: null, + coverage: { ...coverage, ...(over.coverage ?? {}) }, + objects, + issues: over.issues ?? [], + audit: over.audit ?? [], + backupsOmitted: 0, + } +} + +describe('buildCNPGFleet', () => { + it('never reports replication as healthy from pod readiness alone', () => { + const fleet = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db')], + pods: [pod('pg-a-1', 'db', 'pg-a', 'primary'), pod('pg-a-2', 'db', 'pg-a', 'replica'), pod('pg-a-3', 'db', 'pg-a', 'replica')], + }), + ) + const row = fleet.rows[0] + expect(row.replication.tone).toBe('unknown') + expect(row.replication.text).toContain('2/2 replicas ready') + expect(row.pods[0].role).toBe('primary') + }) + + it('attributes child-object issues to their cluster and marks attention', () => { + const fleet = buildCNPGFleet( + resp( + { + clusters: [cluster('pg-a', 'db'), cluster('pg-b', 'db')], + databases: [{ apiVersion: G, kind: 'Database', metadata: { name: 'reporting', namespace: 'db' }, spec: { cluster: { name: 'pg-b' } }, status: { applied: false } }], + }, + { + issues: [ + { id: 'i1', severity: 'warning', kind: 'Database', group: 'postgresql.cnpg.io', namespace: 'db', name: 'reporting', reason: 'CNPGDeclarativeNotApplied', message: 'role "x" does not exist' }, + ], + }, + ), + ) + const b = fleet.rows.find((r) => r.name === 'pg-b')! + const a = fleet.rows.find((r) => r.name === 'pg-a')! + expect(b.attention).toBe(true) + expect(b.categories.has('declarations')).toBe(true) + expect(b.declarations.failed).toBe(1) + expect(a.attention).toBe(false) + expect(fleet.attentionCount).toBe(1) + expect(fleet.categoryCounts.declarations).toBe(1) + expect(fleet.rows[0].name).toBe('pg-b') + }) + + it('treats the no-schedule audit finding as posture, not attention, and words it narrowly', () => { + const fleet = buildCNPGFleet( + resp({ clusters: [cluster('pg-a', 'db')] }, { + audit: [{ checkId: 'cnpgNoDeclarativeBackup', severity: 'warning', kind: 'Cluster', namespace: 'db', name: 'pg-a', message: 'no ScheduledBackup' }], + }), + ) + const row = fleet.rows[0] + expect(row.attention).toBe(false) + expect(row.problems[0].title).toBe('No declarative backup schedule') + expect(row.protection.schedule.text).toBe('No declarative schedule') + }) + + it('says "no access" rather than "none" when a kind is not readable', () => { + const fleet = buildCNPGFleet( + resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'denied' }, scheduledBackups: { state: 'partial', deniedNamespaces: ['db'] } } }), + ) + const p = fleet.rows[0].protection + expect(p.lastSuccessfulBackup.text).toBe('No access to Backups') + expect(p.lastSuccessfulBackup.tone).toBe('unknown') + expect(p.schedule.text).toBe('No access to ScheduledBackups') + expect(fleet.incompleteKinds).toEqual(expect.arrayContaining(['backups', 'scheduledBackups'])) + }) + + it('prefers the newest successful backup across Backup CRs and ObjectStore status, citing the source', () => { + const fleet = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } })], + backups: [{ apiVersion: G, kind: 'Backup', metadata: { name: 'pg-a-old', namespace: 'db' }, spec: { cluster: { name: 'pg-a' } }, status: { phase: 'completed', stoppedAt: '2026-09-20T02:00:00Z' } }], + objectStores: [{ + apiVersion: 'barmancloud.cnpg.io/v1', kind: 'ObjectStore', metadata: { name: 'store', namespace: 'db' }, + status: { serverRecoveryWindow: { 'pg-a': { firstRecoverabilityPoint: '2026-09-01T00:00:00Z', lastSuccessfulBackupTime: '2026-09-28T02:00:00Z' } } }, + }], + }), + ) + const p = fleet.rows[0].protection + expect(p.lastSuccessfulBackup.at).toBe('2026-09-28T02:00:00Z') + expect(p.lastSuccessfulBackup.source).toBe('ObjectStore store status') + expect(p.recoveryWindow.from).toBe('2026-09-01T00:00:00Z') + expect(p.destination.method).toBe('plugin') + }) + + it('ignores deprecated Cluster status backup fields for plugin clusters', () => { + const fleet = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db', { + spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] }, + status: { lastSuccessfulBackup: '2020-01-01T00:00:00Z' }, + })], + }), + ) + expect(fleet.rows[0].protection.lastSuccessfulBackup.text).toBe('None observed') + }) + + it('never marks restore validation healthy', () => { + const src = cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } }) + const restored = cluster('pg-a-restore', 'db', { + spec: { + bootstrap: { recovery: { source: 'origin' } }, + externalClusters: [{ name: 'origin', plugin: { name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store', serverName: 'pg-a' } } }], + }, + }) + const fleet = buildCNPGFleet(resp({ clusters: [src, restored] })) + const a = fleet.rows.find((r) => r.name === 'pg-a')! + expect(a.protection.restoreValidation.text).toBe('Restored into pg-a-restore') + expect(a.protection.restoreValidation.tone).toBe('neutral') + const r = fleet.rows.find((x) => x.name === 'pg-a-restore')! + expect(r.protection.restoreValidation.text).toBe('None recorded') + expect(r.protection.restoreValidation.tone).toBe('unknown') + }) + + it('reports WAL archiving from the condition and unknown when absent', () => { + const failing = cluster('pg-a', 'db', { status: { conditions: [{ type: 'ContinuousArchiving', status: 'False', message: 'exit status 1' }] } }) + const silent = cluster('pg-b', 'db') + const fleet = buildCNPGFleet(resp({ clusters: [failing, silent] })) + expect(fleet.rows.find((r) => r.name === 'pg-a')!.protection.walArchiving.tone).toBe('unhealthy') + expect(fleet.rows.find((r) => r.name === 'pg-a')!.protection.summary.text).toBe('WAL archiving failing') + expect(fleet.rows.find((r) => r.name === 'pg-b')!.protection.walArchiving.tone).toBe('unknown') + }) + + it('ignores same-named kinds from other API groups', () => { + const capi = { apiVersion: 'cluster.x-k8s.io/v1beta1', kind: 'Cluster', metadata: { name: 'workload', namespace: 'db' } } + const fleet = buildCNPGFleet(resp({ clusters: [capi, cluster('pg-a', 'db')] })) + expect(fleet.rows.map((r) => r.name)).toEqual(['pg-a']) + }) + + it('counts declared managed roles and their reconcile errors', () => { + const c = cluster('pg-a', 'db', { + spec: { managed: { roles: [{ name: 'app' }, { name: 'audit' }] } }, + status: { managedRolesStatus: { cannotReconcile: { audit: ['permission denied'] } } }, + }) + const d = buildCNPGFleet(resp({ clusters: [c] })).rows[0].declarations + expect(d.total).toBe(2) + expect(d.failed).toBe(1) + expect(d.summary.tone).toBe('degraded') + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/workspace.ts b/packages/k8s-ui/src/components/cnpg/workspace.ts new file mode 100644 index 000000000..6a332dee9 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/workspace.ts @@ -0,0 +1,646 @@ +// CloudNativePG workspace model: pure derivations over the /api/cnpg/workspace +// payload. Every fact here is something the cluster actually reports; when it +// does not report something the value is "unknown", never zero or healthy. + +import type { HealthLevel } from '../resources/resource-utils' +import { + getCNPGClusterBackupConfig, + getCNPGClusterBarmanPlugin, + getCNPGClusterImageTag, + getCNPGClusterStatus, + getCNPGObjectStoreRecoveryWindows, + isApiGroup, +} from '../resources/resource-utils-cnpg' + +export const CNPG_WORKSPACE_KEYS = [ + 'clusters', + 'backups', + 'scheduledBackups', + 'poolers', + 'databases', + 'publications', + 'subscriptions', + 'imageCatalogs', + 'clusterImageCatalogs', + 'objectStores', + 'pods', +] as const + +export type CNPGWorkspaceKey = (typeof CNPG_WORKSPACE_KEYS)[number] + +export type CNPGCoverageState = 'full' | 'partial' | 'denied' | 'notInstalled' | 'syncing' | 'error' + +export interface CNPGKindCoverage { + state: CNPGCoverageState + deniedNamespaces?: string[] +} + +export interface CNPGWorkspaceIssue { + id: string + severity: 'critical' | 'warning' + category?: string + kind: string + group?: string + namespace?: string + name: string + reason: string + message?: string + cause?: string + action?: string + first_seen?: string +} + +export interface CNPGAuditFinding { + checkId: string + severity: string + kind: string + group?: string + namespace?: string + name: string + message: string +} + +export interface CNPGWorkspaceResponse { + installed: boolean + context: string + namespaces: string[] | null + coverage: Partial> + objects: Partial> + issues: CNPGWorkspaceIssue[] + audit: CNPGAuditFinding[] + backupsOmitted: number +} + +export const CNPG_KIND_BY_KEY: Record = { + clusters: { kind: 'Cluster', group: 'postgresql.cnpg.io', plural: 'clusters' }, + backups: { kind: 'Backup', group: 'postgresql.cnpg.io', plural: 'backups' }, + scheduledBackups: { kind: 'ScheduledBackup', group: 'postgresql.cnpg.io', plural: 'scheduledbackups' }, + poolers: { kind: 'Pooler', group: 'postgresql.cnpg.io', plural: 'poolers' }, + databases: { kind: 'Database', group: 'postgresql.cnpg.io', plural: 'databases' }, + publications: { kind: 'Publication', group: 'postgresql.cnpg.io', plural: 'publications' }, + subscriptions: { kind: 'Subscription', group: 'postgresql.cnpg.io', plural: 'subscriptions' }, + imageCatalogs: { kind: 'ImageCatalog', group: 'postgresql.cnpg.io', plural: 'imagecatalogs' }, + clusterImageCatalogs: { kind: 'ClusterImageCatalog', group: 'postgresql.cnpg.io', plural: 'clusterimagecatalogs' }, + objectStores: { kind: 'ObjectStore', group: 'barmancloud.cnpg.io', plural: 'objectstores' }, + pods: { kind: 'Pod', group: '', plural: 'pods' }, +} + +export function isCNPGWorkspaceKind(kind: string, group: string | undefined): boolean { + return Object.values(CNPG_KIND_BY_KEY).some((k) => k.group !== '' && k.group === (group ?? '') && k.kind === kind) +} + +/** The value is observed, derived, or not available from the cluster. */ +export type CNPGFactTone = HealthLevel + +export interface CNPGFact { + text: string + tone: CNPGFactTone + /** Where the value comes from, shown next to it so claims carry their source. */ + source?: string + /** A timestamp the text refers to; the UI renders it as an age. */ + at?: string +} + +export type CNPGProblemCategory = 'availability' | 'protection' | 'declarations' | 'pooling' + +export const CNPG_PROBLEM_CATEGORIES: { id: CNPGProblemCategory; label: string }[] = [ + { id: 'availability', label: 'Availability' }, + { id: 'protection', label: 'Protection' }, + { id: 'declarations', label: 'Declarations' }, + { id: 'pooling', label: 'Pooling' }, +] + +export interface CNPGProblem { + /** Stable identity for keys. */ + id: string + severity: 'critical' | 'warning' | 'posture' + category: CNPGProblemCategory + title: string + detail?: string + /** The object the evidence is about (may be the Cluster or a child object). */ + subject: { kind: string; group: string; namespace: string; name: string } + source: 'issue' | 'audit' +} + +export interface CNPGInstance { + name: string + role: 'primary' | 'replica' | 'unknown' + ready: boolean | null + node?: string + zone?: string +} + +export interface CNPGProtectionFacts { + schedule: CNPGFact & { names: string[] } + destination: CNPGFact & { + method: 'plugin' | 'barmanObjectStore' | 'volumeSnapshot' | 'none' + objectStore?: string + } + lastSuccessfulBackup: CNPGFact + walArchiving: CNPGFact + recoveryWindow: CNPGFact & { from?: string; to?: string } + restoreValidation: CNPGFact & { restoredInto?: { namespace: string; name: string } } +} + +export interface CNPGFleetRow { + key: string + namespace: string + name: string + cluster: any + controllerStatus: { text: string; level: HealthLevel } + instances: { ready: number | null; desired: number | null } + pods: CNPGInstance[] + replicaCluster: { source?: string } | null + hibernated: boolean + pgVersion: string | null + catalog: { kind: string; name: string } | null + replication: CNPGFact + protection: CNPGProtectionFacts & { summary: CNPGFact } + declarations: { summary: CNPGFact; total: number; failed: number; pending: number } + poolers: string[] + problems: CNPGProblem[] + /** Any issue at warning or worse. Posture findings alone do not need attention. */ + attention: boolean + categories: Set + /** GitOps owner recorded on the Cluster, when it carries the standard labels. */ + gitops: { tool: 'argocd' | 'flux'; name: string; namespace?: string } | null +} + +export interface CNPGFleet { + rows: CNPGFleetRow[] + attentionCount: number + categoryCounts: Record + /** Kinds whose coverage is not complete, so counts built on them are lower bounds. */ + incompleteKinds: CNPGWorkspaceKey[] +} + +const PROTECTION_ISSUE_REASONS = new Set([ + 'CNPGWALArchivingFailing', + 'CNPGLastBackupFailed', + 'CNPGBackupFailed', + 'CNPGScheduledBackupMissed', +]) + +function issueCategory(issue: CNPGWorkspaceIssue): CNPGProblemCategory { + if (PROTECTION_ISSUE_REASONS.has(issue.reason)) return 'protection' + switch (issue.kind) { + case 'Backup': + case 'ScheduledBackup': + case 'ObjectStore': + return 'protection' + case 'Database': + case 'Publication': + case 'Subscription': + return 'declarations' + case 'Pooler': + return 'pooling' + default: + return 'availability' + } +} + +function key(ns: string | undefined, name: string): string { + return `${ns ?? ''}/${name}` +} + +function specClusterName(obj: any): string | undefined { + const n = obj?.spec?.cluster?.name + return typeof n === 'string' && n ? n : undefined +} + +function coverageOf(resp: CNPGWorkspaceResponse, k: CNPGWorkspaceKey): CNPGKindCoverage { + return resp.coverage?.[k] ?? { state: 'notInstalled' } +} + +/** Coverage is usable in a namespace when that namespace's objects were read. */ +export function coverageReadable(cov: CNPGKindCoverage, namespace?: string): boolean { + if (cov.state === 'full') return true + if (cov.state === 'partial') return !namespace || !(cov.deniedNamespaces ?? []).includes(namespace) + return false +} + +function coverageUnavailableText(cov: CNPGKindCoverage, what: string): string { + switch (cov.state) { + case 'denied': + case 'partial': + return `No access to ${what}` + case 'syncing': + return 'Loading…' + case 'error': + return `Could not read ${what}` + default: + return 'Not installed' + } +} + +function timeOf(obj: any): number { + const t = obj?.status?.stoppedAt || obj?.status?.startedAt || obj?.metadata?.creationTimestamp + const ms = t ? Date.parse(t) : NaN + return Number.isFinite(ms) ? ms : 0 +} + +function podRole(pod: any, cluster: any): CNPGInstance['role'] { + const role = pod?.metadata?.labels?.['cnpg.io/instanceRole'] ?? pod?.metadata?.labels?.role + if (role === 'primary') return 'primary' + if (role === 'replica') return 'replica' + const primary = cluster?.status?.currentPrimary + if (primary && pod?.metadata?.name === primary) return 'primary' + return 'unknown' +} + +function podReady(pod: any): boolean | null { + const conds = pod?.status?.conditions + if (!Array.isArray(conds)) return null + const ready = conds.find((c: any) => c?.type === 'Ready') + if (!ready) return null + return ready.status === 'True' +} + +function clusterGitOps(cluster: any): CNPGFleetRow['gitops'] { + const labels = cluster?.metadata?.labels ?? {} + const annotations = cluster?.metadata?.annotations ?? {} + const argo = labels['argocd.argoproj.io/instance'] + if (argo) return { tool: 'argocd', name: argo } + const tracking = annotations['argocd.argoproj.io/tracking-id'] + if (typeof tracking === 'string' && tracking.includes(':')) { + return { tool: 'argocd', name: tracking.split(':')[0] } + } + const fluxName = labels['kustomize.toolkit.fluxcd.io/name'] || labels['helm.toolkit.fluxcd.io/name'] + if (fluxName) { + return { + tool: 'flux', + name: fluxName, + namespace: labels['kustomize.toolkit.fluxcd.io/namespace'] || labels['helm.toolkit.fluxcd.io/namespace'], + } + } + return null +} + +function scheduleFact( + cluster: any, + schedules: any[], + cov: CNPGKindCoverage, +): CNPGProtectionFacts['schedule'] { + const ns = cluster.metadata?.namespace + if (!coverageReadable(cov, ns)) { + return { text: coverageUnavailableText(cov, 'ScheduledBackups'), tone: 'unknown', names: [] } + } + const mine = schedules.filter((s) => s.metadata?.namespace === ns && specClusterName(s) === cluster.metadata?.name) + if (mine.length === 0) return { text: 'No declarative schedule', tone: 'neutral', names: [] } + const active = mine.filter((s) => s.spec?.suspend !== true) + const names = mine.map((s) => s.metadata?.name).filter(Boolean) + if (active.length === 0) { + return { text: mine.length === 1 ? 'Schedule suspended' : 'All schedules suspended', tone: 'degraded', names } + } + const cron = active[0]?.spec?.schedule + return { + text: active.length === 1 ? (cron ? `Scheduled · ${cron}` : 'Scheduled') : `${active.length} schedules`, + tone: 'healthy', + names, + } +} + +function destinationFact(cluster: any): CNPGProtectionFacts['destination'] { + const plugin = getCNPGClusterBarmanPlugin(cluster) + if (plugin?.barmanObjectName) { + return { + text: `ObjectStore ${plugin.barmanObjectName}`, + tone: 'neutral', + method: 'plugin', + objectStore: plugin.barmanObjectName, + } + } + const cfg = getCNPGClusterBackupConfig(cluster) + if (cfg.destinationPath) { + return { text: cfg.destinationPath, tone: 'neutral', method: 'barmanObjectStore' } + } + if (cluster?.spec?.backup?.volumeSnapshot) { + return { text: 'Volume snapshots', tone: 'neutral', method: 'volumeSnapshot' } + } + return { text: 'No destination configured', tone: 'neutral', method: 'none' } +} + +function recoveryWindowFor(cluster: any, stores: any[]): { from?: string; lastSuccess?: string; lastFailed?: string; store?: string } | null { + const plugin = getCNPGClusterBarmanPlugin(cluster) + if (!plugin?.barmanObjectName) return null + const store = stores.find( + (s) => s.metadata?.namespace === cluster.metadata?.namespace && s.metadata?.name === plugin.barmanObjectName, + ) + if (!store) return null + const server = plugin.serverName || cluster.metadata?.name + const w = getCNPGObjectStoreRecoveryWindows(store).find((x) => x.server === server) + if (!w) return { store: store.metadata?.name } + return { + from: w.firstRecoverabilityPoint, + lastSuccess: w.lastSuccessfulBackupTime, + lastFailed: w.lastFailedBackupTime, + store: store.metadata?.name, + } +} + +function lastBackupFact( + cluster: any, + backups: any[], + backupsCov: CNPGKindCoverage, + window: ReturnType, +): CNPGProtectionFacts['lastSuccessfulBackup'] { + const ns = cluster.metadata?.namespace + const name = cluster.metadata?.name + const candidates: { at: string; source: string }[] = [] + if (coverageReadable(backupsCov, ns)) { + const completed = backups + .filter( + (b) => + b.metadata?.namespace === ns && + specClusterName(b) === name && + isApiGroup(b.apiVersion, 'postgresql.cnpg.io') && + b.status?.phase === 'completed', + ) + .sort((a, b) => timeOf(b) - timeOf(a))[0] + const at = completed?.status?.stoppedAt || completed?.status?.startedAt + if (completed && at) candidates.push({ at, source: `Backup ${completed.metadata?.name}` }) + } + if (window?.lastSuccess) candidates.push({ at: window.lastSuccess, source: `ObjectStore ${window.store} status` }) + const cfg = getCNPGClusterBackupConfig(cluster) + if (!cfg.plugin && cfg.lastSuccessfulBackup) candidates.push({ at: cfg.lastSuccessfulBackup, source: 'Cluster status' }) + if (candidates.length === 0) { + if (!coverageReadable(backupsCov, ns)) { + return { text: coverageUnavailableText(backupsCov, 'Backups'), tone: 'unknown' } + } + return { text: 'None observed', tone: 'unknown' } + } + const best = candidates.reduce((a, b) => (Date.parse(a.at) >= Date.parse(b.at) ? a : b)) + return { text: 'Completed', tone: 'healthy', at: best.at, source: best.source } +} + +function walFact(cluster: any): CNPGFact { + const conds = cluster?.status?.conditions + const c = Array.isArray(conds) ? conds.find((x: any) => x?.type === 'ContinuousArchiving') : null + if (!c) return { text: 'Not reported', tone: 'unknown', source: 'Cluster status' } + if (c.status === 'True') return { text: 'Archiving', tone: 'healthy', source: 'ContinuousArchiving condition' } + if (c.status === 'False') { + return { text: c.message ? `Failing · ${c.message}` : 'Failing', tone: 'unhealthy', source: 'ContinuousArchiving condition' } + } + return { text: 'Unknown', tone: 'unknown', source: 'ContinuousArchiving condition' } +} + +function restoreValidationFact( + cluster: any, + allClusters: any[], +): CNPGProtectionFacts['restoreValidation'] { + const plugin = getCNPGClusterBarmanPlugin(cluster) + const server = plugin?.serverName || cluster.metadata?.name + const store = plugin?.barmanObjectName + const ns = cluster.metadata?.namespace + const restored = allClusters.find((c) => { + if (c === cluster || c.metadata?.namespace !== ns) return false + const recovery = c.spec?.bootstrap?.recovery + if (!recovery) return false + const sourceName = recovery.source + const ext = (c.spec?.externalClusters ?? []).find((e: any) => e?.name === sourceName) + const params = ext?.plugin?.parameters + if (store && params?.barmanObjectName === store && (params?.serverName || sourceName) === server) return true + const backupName = recovery.backup?.name + return !!backupName && typeof backupName === 'string' && backupName.startsWith(`${cluster.metadata?.name}-`) + }) + if (restored) { + return { + text: `Restored into ${restored.metadata?.name}`, + tone: 'neutral', + source: `Cluster ${restored.metadata?.name} bootstrapped from this cluster's backups · created ${restored.metadata?.creationTimestamp ?? 'unknown'}`, + restoredInto: { namespace: restored.metadata?.namespace, name: restored.metadata?.name }, + } + } + return { text: 'None recorded', tone: 'unknown', source: 'Kubernetes does not record restore tests' } +} + +function protectionSummary(p: CNPGProtectionFacts): CNPGFact { + if (p.walArchiving.tone === 'unhealthy') return { text: 'WAL archiving failing', tone: 'unhealthy' } + if (p.destination.method === 'none' && p.schedule.names.length === 0) { + return { text: 'No backup destination or schedule', tone: 'neutral' } + } + if (p.schedule.tone === 'degraded') return { text: p.schedule.text, tone: 'degraded' } + if (p.lastSuccessfulBackup.at) return { text: 'Last backup', tone: 'healthy', at: p.lastSuccessfulBackup.at } + return { text: p.lastSuccessfulBackup.text, tone: p.lastSuccessfulBackup.tone } +} + +function catalogRef(cluster: any): CNPGFleetRow['catalog'] { + const ref = cluster?.spec?.imageCatalogRef + if (!ref?.name) return null + return { kind: ref.kind || 'ImageCatalog', name: ref.name } +} + +function pgVersion(cluster: any): string | null { + const tag = getCNPGClusterImageTag(cluster) + if (tag && tag !== '-') { + const m = tag.match(/^(\d+(?:\.\d+)?)/) + if (m) return m[1] + } + const major = cluster?.spec?.imageCatalogRef?.major + return typeof major === 'number' ? String(major) : null +} + +function replicationFact(cluster: any, pods: CNPGInstance[], hibernated: boolean): CNPGFact { + if (hibernated) return { text: 'Hibernated', tone: 'neutral' } + const desired = cluster?.spec?.instances + if (desired === 1) return { text: 'Single instance', tone: 'neutral' } + const replicas = pods.filter((p) => p.role === 'replica') + const readyReplicas = replicas.filter((p) => p.ready === true).length + if (replicas.length === 0) return { text: 'No replica pods observed', tone: 'unknown' } + return { + text: `${readyReplicas}/${replicas.length} replicas ready · lag unknown`, + tone: 'unknown', + source: 'Pod readiness does not show whether a replica is streaming', + } +} + +function problemsFor( + cluster: any, + issues: CNPGWorkspaceIssue[], + audit: CNPGAuditFinding[], + children: Map, +): CNPGProblem[] { + const ns = cluster.metadata?.namespace + const name = cluster.metadata?.name + const out: CNPGProblem[] = [] + for (const issue of issues) { + if ((issue.namespace ?? '') !== ns) continue + const isSelf = issue.kind === 'Cluster' && issue.name === name + const owner = children.get(`${issue.kind}/${ns}/${issue.name}`) + if (!isSelf && owner !== name) continue + out.push({ + id: issue.id, + severity: issue.severity, + category: issueCategory(issue), + title: issue.message || issue.reason, + detail: issue.cause || undefined, + subject: { kind: issue.kind, group: issue.group ?? '', namespace: ns, name: issue.name }, + source: 'issue', + }) + } + for (const f of audit) { + if (f.kind !== 'Cluster' || f.name !== name || (f.namespace ?? '') !== ns) continue + out.push({ + id: `audit:${f.checkId}:${ns}/${name}`, + severity: 'posture', + category: 'protection', + title: f.checkId === 'cnpgNoDeclarativeBackup' ? 'No declarative backup schedule' : f.message, + detail: f.message, + subject: { kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: ns, name }, + source: 'audit', + }) + } + const rank = { critical: 0, warning: 1, posture: 2 } as const + return out.sort((a, b) => rank[a.severity] - rank[b.severity] || a.title.localeCompare(b.title)) +} + +/** Index "Kind/ns/name" → owning cluster name, from each child's spec.cluster.name. */ +function childIndex(resp: CNPGWorkspaceResponse): Map { + const idx = new Map() + const add = (kind: string, list: any[] | undefined) => { + for (const o of list ?? []) { + const c = specClusterName(o) + if (c) idx.set(`${kind}/${o.metadata?.namespace}/${o.metadata?.name}`, c) + } + } + add('Backup', resp.objects.backups) + add('ScheduledBackup', resp.objects.scheduledBackups) + add('Pooler', resp.objects.poolers) + add('Database', resp.objects.databases) + add('Publication', resp.objects.publications) + add('Subscription', resp.objects.subscriptions) + return idx +} + +function declarationsFor(cluster: any, resp: CNPGWorkspaceResponse): CNPGFleetRow['declarations'] { + const ns = cluster.metadata?.namespace + const name = cluster.metadata?.name + const lists: [CNPGWorkspaceKey, any[]][] = [ + ['databases', resp.objects.databases ?? []], + ['publications', resp.objects.publications ?? []], + ['subscriptions', resp.objects.subscriptions ?? []], + ] + let total = 0 + let failed = 0 + let pending = 0 + let unreadable = false + for (const [k, list] of lists) { + if (!coverageReadable(coverageOf(resp, k), ns)) { + if (coverageOf(resp, k).state !== 'notInstalled') unreadable = true + continue + } + for (const o of list) { + if (o.metadata?.namespace !== ns || specClusterName(o) !== name) continue + total++ + if (o.status?.applied === false) failed++ + else if (o.status?.applied !== true) pending++ + } + } + const roles = cluster?.status?.managedRolesStatus + const roleErrors = roles?.cannotReconcile ? Object.keys(roles.cannotReconcile).length : 0 + const declaredRoles = Array.isArray(cluster?.spec?.managed?.roles) ? cluster.spec.managed.roles.length : 0 + failed += roleErrors + total += declaredRoles + let summary: CNPGFact + if (total === 0) { + summary = unreadable ? { text: 'No access to some declarations', tone: 'unknown' } : { text: 'None declared', tone: 'neutral' } + } else if (failed > 0) { + summary = { text: `${failed} of ${total} not reconciled`, tone: 'degraded' } + } else if (pending > 0) { + summary = { text: `${pending} of ${total} pending`, tone: 'unknown' } + } else { + summary = { text: `${total} reconciled`, tone: 'healthy' } + } + if (unreadable && total > 0) summary = { ...summary, source: 'Some declaration kinds are not readable' } + return { summary, total, failed, pending } +} + +export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { + const clusters = (resp.objects.clusters ?? []).filter((c) => isApiGroup(c?.apiVersion, 'postgresql.cnpg.io')) + const pods = resp.objects.pods ?? [] + const stores = resp.objects.objectStores ?? [] + const children = childIndex(resp) + const poolers = resp.objects.poolers ?? [] + + const rows: CNPGFleetRow[] = clusters.map((cluster) => { + const ns = cluster.metadata?.namespace ?? '' + const name = cluster.metadata?.name ?? '' + const status = getCNPGClusterStatus(cluster) + const hibernated = cluster?.metadata?.annotations?.['cnpg.io/hibernation'] === 'on' + const instancePods: CNPGInstance[] = pods + .filter((p) => p.metadata?.namespace === ns && p.metadata?.labels?.['cnpg.io/cluster'] === name) + .map((p) => ({ + name: p.metadata?.name, + role: podRole(p, cluster), + ready: podReady(p), + node: p.spec?.nodeName, + })) + .sort((a, b) => (a.role === 'primary' ? -1 : b.role === 'primary' ? 1 : a.name.localeCompare(b.name))) + const readyInstances = typeof cluster?.status?.readyInstances === 'number' ? cluster.status.readyInstances : null + const desired = typeof cluster?.spec?.instances === 'number' ? cluster.spec.instances : null + + const window = recoveryWindowFor(cluster, stores) + const protection: CNPGProtectionFacts = { + schedule: scheduleFact(cluster, resp.objects.scheduledBackups ?? [], coverageOf(resp, 'scheduledBackups')), + destination: destinationFact(cluster), + lastSuccessfulBackup: lastBackupFact(cluster, resp.objects.backups ?? [], coverageOf(resp, 'backups'), window), + walArchiving: walFact(cluster), + recoveryWindow: window?.from + ? { + text: 'Recoverable window', + tone: window.lastFailed && (!window.lastSuccess || Date.parse(window.lastFailed) > Date.parse(window.lastSuccess)) ? 'degraded' : 'neutral', + from: window.from, + to: window.lastSuccess, + source: `ObjectStore ${window.store} status`, + } + : { text: 'Not reported', tone: 'unknown' }, + restoreValidation: restoreValidationFact(cluster, clusters), + } + const problems = problemsFor(cluster, resp.issues ?? [], resp.audit ?? [], children) + const categories = new Set( + problems.filter((p) => p.severity !== 'posture').map((p) => p.category), + ) + const replica = cluster?.spec?.replica?.enabled ? { source: cluster.spec.replica.source } : null + + return { + key: key(ns, name), + namespace: ns, + name, + cluster, + controllerStatus: { text: status.text, level: status.level }, + instances: { ready: readyInstances, desired }, + pods: instancePods, + replicaCluster: replica, + hibernated, + pgVersion: pgVersion(cluster), + catalog: catalogRef(cluster), + replication: replicationFact(cluster, instancePods, hibernated), + protection: { ...protection, summary: protectionSummary(protection) }, + declarations: declarationsFor(cluster, resp), + poolers: poolers + .filter((p) => p.metadata?.namespace === ns && specClusterName(p) === name) + .map((p) => p.metadata?.name), + problems, + attention: problems.some((p) => p.severity !== 'posture'), + categories, + gitops: clusterGitOps(cluster), + } + }) + + rows.sort((a, b) => Number(b.attention) - Number(a.attention) || a.namespace.localeCompare(b.namespace) || a.name.localeCompare(b.name)) + + const categoryCounts = { availability: 0, protection: 0, declarations: 0, pooling: 0 } as Record + for (const r of rows) for (const c of r.categories) categoryCounts[c]++ + + const incompleteKinds = CNPG_WORKSPACE_KEYS.filter((k) => { + const s = coverageOf(resp, k).state + return s === 'partial' || s === 'denied' || s === 'syncing' || s === 'error' + }) + + return { + rows, + attentionCount: rows.filter((r) => r.attention).length, + categoryCounts, + incompleteKinds, + } +} diff --git a/packages/k8s-ui/src/components/resources/ResourcesSidebar.test.tsx b/packages/k8s-ui/src/components/resources/ResourcesSidebar.test.tsx index 39c0c9f46..f76df8d7f 100644 --- a/packages/k8s-ui/src/components/resources/ResourcesSidebar.test.tsx +++ b/packages/k8s-ui/src/components/resources/ResourcesSidebar.test.tsx @@ -105,3 +105,79 @@ describe('ResourcesSidebar count visibility', () => { expect(html).toContain('HorizontalPodAutoscaler') }) }) + +describe('ResourcesSidebar category workspaces', () => { + const cnpgCluster: APIResource = { + group: 'postgresql.cnpg.io', + version: 'v1', + kind: 'Cluster', + name: 'clusters', + namespaced: true, + isCrd: true, + verbs: ['list'], + } + const objectStore: APIResource = { + group: 'barmancloud.cnpg.io', + version: 'v1', + kind: 'ObjectStore', + name: 'objectstores', + namespaced: true, + isCrd: true, + verbs: ['list'], + } + + it('renders destinations above the kinds, with the active object nested and no kind selected', () => { + const html = renderToString( + {}} + apiResources={[cnpgCluster, objectStore]} + resourceCounts={{ 'postgresql.cnpg.io/Cluster': 2, 'barmancloud.cnpg.io/ObjectStore': 1 }} + categoryWorkspaces={{ + CloudNativePG: { + destinations: [ + { id: 'overview', label: 'Overview', count: 3, countTitle: '3 clusters need attention', active: true, child: { label: 'pg-orders' }, onSelect: () => {} }, + ], + defaultKindsCollapsed: true, + scopeNote: 'Counts for namespace payments', + }, + }} + /> + ) + expect(html).toContain('Workspace') + expect(html).toContain('Overview') + expect(html).toContain('pg-orders') + expect(html).toContain('Counts for namespace payments') + expect(html).toContain('Resource kinds') + expect(html).toMatch(/aria-expanded="false"[^>]*>(?:(?!<\/button>).)*Resource kinds/) + expect(html).not.toContain('selection-strong selection-text">Pod') + }) + + it('keeps a workspace category visible when it has no resources', () => { + const html = renderToString( + {}} + apiResources={[cnpgCluster]} + resourceCounts={{ 'postgresql.cnpg.io/Cluster': 0 }} + categoryWorkspaces={{ CloudNativePG: { destinations: [{ id: 'overview', label: 'Overview', onSelect: () => {} }] } }} + /> + ) + expect(html).toContain('CloudNativePG') + expect(html).toContain('Overview') + }) + + it('labels API groups when a workspace category spans several', () => { + const html = renderToString( + {}} + apiResources={[cnpgCluster, objectStore]} + resourceCounts={{ 'postgresql.cnpg.io/Cluster': 2, 'barmancloud.cnpg.io/ObjectStore': 1 }} + categoryWorkspaces={{ CloudNativePG: { destinations: [{ id: 'overview', label: 'Overview', onSelect: () => {} }], defaultKindsCollapsed: true } }} + /> + ) + expect(html).toContain('postgresql.cnpg.io') + expect(html).toContain('barmancloud.cnpg.io') + }) +}) diff --git a/packages/k8s-ui/src/components/resources/ResourcesSidebar.tsx b/packages/k8s-ui/src/components/resources/ResourcesSidebar.tsx index 792a03ee8..c06bc32e2 100644 --- a/packages/k8s-ui/src/components/resources/ResourcesSidebar.tsx +++ b/packages/k8s-ui/src/components/resources/ResourcesSidebar.tsx @@ -1,4 +1,4 @@ -import { useState, useMemo, useEffect, useRef, useCallback, useId, forwardRef } from 'react' +import { useState, useMemo, useEffect, useRef, useCallback, useId, forwardRef, type ComponentType } from 'react' import { Search, Eye, @@ -29,6 +29,28 @@ export interface PinnedItem { group: string } +/** A task destination shown inside a category, above its exact kinds. */ +export interface SidebarCategoryDestination { + id: string + label: string + icon?: ComponentType<{ className?: string }> + /** Problem count. `undefined` renders no badge; `null` renders the unknown dash. */ + count?: number | null + countTitle?: string + active?: boolean + /** The object currently open under this destination, nested beneath it. */ + child?: { label: string; title?: string } + onSelect: () => void +} + +export interface SidebarCategoryWorkspace { + destinations: SidebarCategoryDestination[] + /** Kinds start collapsed under a "Resource kinds" disclosure until the user opens them. */ + defaultKindsCollapsed?: boolean + /** Explains what the destination counts are scoped to. */ + scopeNote?: string +} + export interface ResourcesSidebarProps { selectedKind: SelectedKindInfo | null onSelectedKindChange: (kind: SelectedKindInfo) => void @@ -48,10 +70,15 @@ export interface ResourcesSidebarProps { /** Called when a kind is selected via keyboard (Enter in the filter). Parent uses this * to move focus to the next UI level (e.g., the table search input). */ onKindNavigated?: () => void + /** Task destinations keyed by category name (e.g. "CloudNativePG"). A category + * with a workspace stays visible even when it has no resources. */ + categoryWorkspaces?: Record } // Persisted across remounts so collapsed categories survive tab switches let persistedExpandedCategories: Set | null = null +// Per-category "Resource kinds" disclosure, once the user has toggled it. +const persistedKindsOpen = new Map() const COUNT_UNAVAILABLE_MESSAGE = 'Count unavailable. Open to view resources.' // Fallback resource types when API resources aren't loaded yet @@ -102,10 +129,11 @@ interface ResourceTypeButtonProps { isPinned?: boolean onTogglePin?: () => void onClick: () => void + indent?: boolean } const ResourceTypeButton = forwardRef( - function ResourceTypeButton({ resource, count, isSelected, isHighlighted, isForbidden: forbidden, isPinned, onTogglePin, onClick }, ref) { + function ResourceTypeButton({ resource, count, isSelected, isHighlighted, isForbidden: forbidden, isPinned, onTogglePin, onClick, indent }, ref) { const Icon = getResourceIcon(resource.kind, resource.group) return ( + {d.child && ( +
+ {d.child.label} +
+ )} + + ) + })} + {workspace.scopeNote && ( +
{workspace.scopeNote}
+ )} + + + ) +} diff --git a/packages/k8s-ui/src/components/resources/ResourcesView.tsx b/packages/k8s-ui/src/components/resources/ResourcesView.tsx index b4cfafed1..a6f4a8027 100644 --- a/packages/k8s-ui/src/components/resources/ResourcesView.tsx +++ b/packages/k8s-ui/src/components/resources/ResourcesView.tsx @@ -205,7 +205,7 @@ import { CalicoInfraCell, CalicoPolicyCell } from './renderers/calico-cells' import { isCalicoPolicyResource, isCoreNetworkPolicyKind } from './resource-utils-calico' import { useRegisterShortcut, useRegisterShortcuts } from '../../hooks/useKeyboardShortcuts' import { ResourcesSidebar } from './ResourcesSidebar' -import type { SelectedKindInfo } from './ResourcesSidebar' +import type { SelectedKindInfo, SidebarCategoryWorkspace } from './ResourcesSidebar' import { CompareTray, togglePick, pickIndex, refToParam, SIDE_TONES, type CompareTrayPick, type NamespacedRef } from '../compare' import { ConfirmDialog } from '../ui/ConfirmDialog' @@ -3411,6 +3411,8 @@ interface ResourcesViewProps { onSelectedKindChange?: (kind: { name: string; kind: string; group: string }) => void /** When true, the sidebar is not rendered. Useful when a standalone ResourcesSidebar is used externally. */ hideSidebar?: boolean + /** Task destinations rendered inside sidebar categories (see ResourcesSidebar). */ + sidebarCategoryWorkspaces?: Record /** Callback when the [+] create button is clicked. Receives the currently selected kind info. */ onCreateResource?: (kind: { name: string; kind: string; group: string } | null) => void /** Default kind when the URL does not include one. */ @@ -3653,6 +3655,7 @@ export function ResourcesView({ onOpenWorkloadLogs, onSelectedKindChange, hideSidebar = false, + sidebarCategoryWorkspaces, onCreateResource, defaultKind = DEFAULT_KIND_INFO, extraLeadingColumns, @@ -5867,6 +5870,7 @@ export function ResourcesView({ pinned={pinned} togglePin={togglePin} isPinned={isPinned} + categoryWorkspaces={sidebarCategoryWorkspaces} onKindNavigated={() => { // After selecting a kind via keyboard, move focus to the table search // so the user can immediately filter within the selected kind. diff --git a/packages/k8s-ui/src/components/resources/index.ts b/packages/k8s-ui/src/components/resources/index.ts index f872236ff..ae7ff0a22 100644 --- a/packages/k8s-ui/src/components/resources/index.ts +++ b/packages/k8s-ui/src/components/resources/index.ts @@ -54,7 +54,7 @@ export * from './resource-utils-velero' export { ResourcesView, ResourcesViewDataContext, hasCuratedColumns } from './ResourcesView' export type { ResourceQueryResult, ExtraColumn, LargeListGuardState } from './ResourcesView' export { ResourcesSidebar } from './ResourcesSidebar' -export type { ResourcesSidebarProps, SelectedKindInfo, PinnedItem } from './ResourcesSidebar' +export type { ResourcesSidebarProps, SelectedKindInfo, PinnedItem, SidebarCategoryDestination, SidebarCategoryWorkspace } from './ResourcesSidebar' export { sanitizePrinterTable, printerTableKey, diff --git a/packages/k8s-ui/src/components/workload/WorkloadView.tsx b/packages/k8s-ui/src/components/workload/WorkloadView.tsx index 8e6b83f72..248d635a6 100644 --- a/packages/k8s-ui/src/components/workload/WorkloadView.tsx +++ b/packages/k8s-ui/src/components/workload/WorkloadView.tsx @@ -77,7 +77,7 @@ import { rolloutMayAdvanceAutomatically, type WorkloadRolloutActivity } from '.. import { WorkloadRolloutNotice } from './WorkloadRolloutNotice' import { isCoreBatchJob } from '../../utils/api-resources' -export type WorkloadTabType = 'overview' | 'topology' | 'timeline' | 'logs' | 'metrics' | 'reachability' | 'cost' | 'yaml' +export type WorkloadTabType = 'overview' | 'spec' | 'topology' | 'timeline' | 'logs' | 'metrics' | 'reachability' | 'cost' | 'yaml' type TabType = WorkloadTabType export interface ResourceOwnershipContext { @@ -286,6 +286,22 @@ interface WorkloadViewProps { initialContainer: string | null onConsumeInitialContainer: () => void }) => ReactNode + /** + * A composed summary for kinds that have one. When it returns content, the + * Overview tab shows the summary and the resource's own renderer moves to a + * "Spec & status" tab, so the same facts never render twice. Return null to + * keep the default Overview. + */ + renderSummary?: (props: { + kind: string + apiKind: string + namespace: string + name: string + group?: string + resource: any + context: 'drawer' | 'expanded' + onNavigate?: NavigateToResource + }) => ReactNode /** Render a full replacement for the expanded Overview tab. */ renderExpandedOverview?: (props: { kind: string @@ -434,6 +450,7 @@ export function WorkloadView({ renderDiagnoseTab, reachableVia, renderExpandedOverview, + renderSummary, renderRelatedYaml, renderMetricsTab, renderCostTab, @@ -480,8 +497,10 @@ export function WorkloadView({ // Collapsed mode state (YAML toggle for drawer mode) const [showYaml, setShowYaml] = useState(initialTab === 'yaml') + const [drawerSpec, setDrawerSpec] = useState(false) useEffect(() => { setShowYaml(initialTab === 'yaml') + setDrawerSpec(false) }, [kindProp, namespace, name, initialTab]) const switchView = useCallback((yaml: boolean) => { @@ -776,8 +795,12 @@ export function WorkloadView({ const podEvidenceLoading = resourceLoading || workloadPodsLoading || eventsLoading const logsFallbackReady = !renderLogsTab || (!logsTabVisible && !podEvidenceLoading) const requestedTab: TabType = activeTab + const summaryContext = { kind, apiKind, namespace, name, group, resource, onNavigate: onNavigateToResource } + const expandedSummary = expanded && resource ? renderSummary?.({ ...summaryContext, context: 'expanded' }) ?? null : null + const drawerSummary = !expanded && resource ? renderSummary?.({ ...summaryContext, context: 'drawer' }) ?? null : null const tabs: DetailShellTab[] = [ { id: 'overview', label: 'Overview', icon: }, + { id: 'spec', label: 'Spec & status', icon: , hidden: !expandedSummary }, { id: 'topology', label: 'Topology', icon: , hidden: topologyTabHidden }, { id: 'timeline', @@ -802,6 +825,7 @@ export function WorkloadView({ requestedTab !== 'overview' && !requestedTabAvailable && ( + (requestedTab === 'spec' && !!resource && !resourceLoading && !expandedSummary) || (requestedTab === 'topology' && topologyTabHidden) || (requestedTab === 'metrics' && (!renderMetricsTab || (!!resource && !resourceLoading && !showMetricsTab))) || (requestedTab === 'cost' && (!renderCostTab || (!!resource && !resourceLoading && !showCostTab))) || @@ -941,6 +965,33 @@ export function WorkloadView({ /> ) : ( + {drawerSummary && ( +
+
+ {([['overview', 'Overview'], ['spec', 'Spec & status']] as const).map(([id, label]) => { + const on = id === 'spec' ? drawerSpec : !drawerSpec + return ( + + ) + })} +
+
+ )} + {drawerSummary && !drawerSpec ? ( + drawerSummary + ) : ( + <> {renderOverviewLead && hasOperationalIssues && (
{renderOverviewLead({ kind, namespace, name })} @@ -969,6 +1020,8 @@ export function WorkloadView({ updatesError={resourceFocusedUpdatesError} mainFooter={renderOverviewExtra && renderOverviewExtra({ kind, namespace, name, group, context: 'drawer' })} /> + + )} )}
@@ -1100,7 +1153,9 @@ export function WorkloadView({ )}
- {effectiveTab === 'overview' && expandedOverview ? ( + {effectiveTab === 'overview' && expandedSummary ? ( +
{expandedSummary}
+ ) : effectiveTab === 'overview' && expandedOverview ? (
{hasOperationalIssues && renderOverviewLead && (
@@ -1109,7 +1164,7 @@ export function WorkloadView({ )} {expandedOverview}
- ) : effectiveTab === 'overview' && ( + ) : (effectiveTab === 'overview' || effectiveTab === 'spec') && ( ([ // Convert API resource name back to topology node ID prefix // Extended MainView type that includes traffic and cost -type ExtendedMainView = MainView | 'traffic' | 'cost' | 'capacity' | 'workload' | 'checks' | 'gitops' | 'compare' | 'helmCompare' | 'issues' | 'applications' | 'investigations' +type ExtendedMainView = MainView | 'traffic' | 'cost' | 'capacity' | 'cnpg' | 'workload' | 'checks' | 'gitops' | 'compare' | 'helmCompare' | 'issues' | 'applications' | 'investigations' // Extract view from URL path function getViewFromPath(pathname: string): ExtendedMainView { @@ -137,6 +138,7 @@ function getViewFromPath(pathname: string): ExtendedMainView { if (path === 'traffic') return 'traffic' if (path === 'cost') return 'cost' if (path === 'capacity') return 'capacity' + if (path === 'cnpg') return 'cnpg' if (path === 'workload') return 'workload' if (path === 'checks' || path === 'audit') return 'checks' // /audit = legacy → checks if (path === 'gitops') return 'gitops' @@ -164,7 +166,7 @@ function usageView(pathname: string, view: ExtendedMainView, upgrade: boolean): const CRASH_LABELS: Record = { home: 'Home', topology: 'Topology', resources: 'Resources', timeline: 'Timeline', issues: 'Issues', helm: 'Helm', helmCompare: 'HelmCompare', traffic: 'Traffic', - cost: 'Cost', capacity: 'Capacity', checks: 'Checks', gitops: 'GitOps', + cost: 'Cost', capacity: 'Capacity', cnpg: 'CloudNativePG', checks: 'Checks', gitops: 'GitOps', applications: 'Applications', workload: 'Workload', compare: 'Compare', investigations: 'Investigations', } @@ -281,6 +283,7 @@ function radarPageTitle(pathname: string, search = '', apiResources?: APIResourc if (pathSegments[1] === 'activity') return 'Capacity Activity' } + if (view === 'cnpg') return 'CloudNativePG' if (view === 'home') return 'Overview' // Every other view's label is its id capitalized — getViewFromPath has already // normalized aliases (e.g. /audit → 'checks'), so no lookup table is needed. @@ -874,7 +877,7 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL gitops: 'g o', checks: 'g u', cost: 'g c', capacity: 'g p', // Non-rail views (reachable via deep links / actions, not the rail) get no // dedicated mnemonic — listed for exhaustiveness so the type stays total. - workload: '', compare: '', helmCompare: '', investigations: '', + workload: '', compare: '', helmCompare: '', investigations: '', cnpg: '', } const views = Object.keys(VIEW_SHORTCUT_KEYS).filter( (v): v is ExtendedMainView => VIEW_SHORTCUT_KEYS[v as ExtendedMainView] !== '', @@ -1063,6 +1066,7 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL secretsChanged: boolean timer: number | null }>({ changedKinds: new Set(), structuralKinds: new Set(), environmentNamespaces: new Set(), environmentPods: new Map(), secretsChanged: false, timer: null }) + const cnpgInvalidationPendingRef = useRef(false) const slowInvalidationRef = useRef<{ updatedKinds: Set // update-only churn → throttled list + dashboard timer: number | null @@ -1094,6 +1098,8 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL const structural = event.operation === 'add' || event.operation === 'delete' const applicationWorkload = ['deployments', 'statefulsets', 'daemonsets', 'rollouts'].includes(kind) + if (event.group?.endsWith('.cnpg.io')) cnpgInvalidationPendingRef.current = true + const fast = fastInvalidationRef.current fast.changedKinds.add(kind) if (structural) fast.structuralKinds.add(kind) @@ -1137,6 +1143,10 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL // GitOps view is mounted (Phase 2 will make this relevance-aware). queryClient.invalidateQueries({ queryKey: ['gitops-tree'] }) queryClient.invalidateQueries({ queryKey: ['gitops-insights'] }) + if (cnpgInvalidationPendingRef.current) { + cnpgInvalidationPendingRef.current = false + queryClient.invalidateQueries({ queryKey: ['cnpg', 'workspace'] }) + } fastInvalidationRef.current = { changedKinds: new Set(), structuralKinds: new Set(), environmentNamespaces: new Set(), environmentPods: new Map(), secretsChanged: false, timer: null } }, 3000) } @@ -1748,7 +1758,7 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL setVisibleKinds(new Set()) }, []) - const navActiveView = mainView === 'helmCompare' ? 'helm' : mainView + const navActiveView = mainView === 'helmCompare' ? 'helm' : mainView === 'cnpg' ? 'resources' : mainView return ( @@ -2300,6 +2310,16 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL )} + {!viewsSyncGated && mainView === 'cnpg' && ( + setSelectedResource(null)} + onClearNamespaces={clearAllNamespaces} + /> + )} + {/* Takeover splash. When the host claims the current view via fleetTakeoverHref, the redirect effect above is mid-flight — render a brief splash instead of the inline view (which would flash + fire its diff --git a/web/src/api/cnpg.ts b/web/src/api/cnpg.ts new file mode 100644 index 000000000..d73971493 --- /dev/null +++ b/web/src/api/cnpg.ts @@ -0,0 +1,21 @@ +import { useQuery } from '@tanstack/react-query' +import type { CNPGWorkspaceResponse } from '@skyhook-io/k8s-ui' +import { fetchJSON } from './client' + +// /api/cnpg/workspace +// +// Every CloudNativePG object the caller may read, with per-kind coverage, the +// CNPG issues on them and the no-schedule audit finding. One query feeds the +// workspace screens, the Resources sidebar counts and the composed summaries, +// so they can never disagree about what they count. +export function useCNPGWorkspace(namespaces: string[], options?: { enabled?: boolean }) { + const ns = [...namespaces].sort().join(',') + return useQuery({ + queryKey: ['cnpg', 'workspace', ns], + queryFn: ({ signal }) => fetchJSON(`/cnpg/workspace${ns ? `?namespaces=${encodeURIComponent(ns)}` : ''}`, signal), + enabled: options?.enabled ?? true, + staleTime: 10_000, + refetchInterval: 30_000, + placeholderData: (prev) => prev, + }) +} diff --git a/web/src/components/cnpg/CNPGOverview.tsx b/web/src/components/cnpg/CNPGOverview.tsx new file mode 100644 index 000000000..d1160adde --- /dev/null +++ b/web/src/components/cnpg/CNPGOverview.tsx @@ -0,0 +1,374 @@ +import { useMemo } from 'react' +import { useNavigate } from 'react-router-dom' +import type { UseQueryResult } from '@tanstack/react-query' +import { clsx } from 'clsx' +import { ArrowRight, Database, Search, X } from 'lucide-react' +import { + CNPG_KIND_BY_KEY, + CNPG_PROBLEM_CATEGORIES, + FactValue, + PaneLoader, + StatusDot, + Tooltip, + toneTextClass, + type CNPGFleet, + type CNPGFleetRow, + type CNPGProblemCategory, + type CNPGWorkspaceResponse, +} from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { useConnection } from '../../context/ConnectionContext' +import { EmptyState, Notice, ROW_HOVER, TABLE_HEAD, TABLE_WRAP, TBODY, TD, TH } from '../capacity/shared' +import { cnpgClusterFullPath } from './paths' +import { sameResource } from './routes' + +const COVERAGE_LABEL: Record = { + denied: 'no access', + partial: 'no access in some namespaces', + syncing: 'still loading', + error: 'could not be read', +} + +type Filter = 'attention' | 'all' + +export function CNPGWorkspaceHeader({ + title, + subtitle, + actions, +}: { + title: string + subtitle?: React.ReactNode + actions?: React.ReactNode +}) { + return ( +
+
+ + CloudNativePG +
+
+
+

{title}

+ {subtitle &&
{subtitle}
} +
+ {actions} +
+
+ ) +} + +export function CoverageNotice({ fleet, data }: { fleet: CNPGFleet; data: CNPGWorkspaceResponse }) { + if (fleet.incompleteKinds.length === 0) return null + const parts = fleet.incompleteKinds.map((k) => { + const cov = data.coverage[k] + const label = COVERAGE_LABEL[cov?.state ?? ''] ?? cov?.state + return `${CNPG_KIND_BY_KEY[k].kind} (${label})` + }) + return ( + + Some CloudNativePG data is not readable: {parts.join(', ')}. Facts built on it read “No access” or “unknown” rather than none, and counts are lower bounds. + + ) +} + +function InstancePills({ row }: { row: CNPGFleetRow }) { + if (row.pods.length === 0) return null + return ( +
+ {row.pods.map((p) => { + const tone = p.ready === true ? 'healthy' : p.ready === false ? 'unhealthy' : 'unknown' + return ( + + + + {p.role === 'primary' ? 'P' : p.role === 'replica' ? 'R' : '?'} + + + ) + })} +
+ ) +} + +function AttentionCell({ row }: { row: CNPGFleetRow }) { + const top = row.problems.find((p) => p.severity !== 'posture') ?? row.problems[0] + if (!top) return — + const tone = top.severity === 'critical' ? 'unhealthy' : top.severity === 'warning' ? 'degraded' : 'neutral' + const more = row.problems.length - 1 + return ( +
+
+ {top.title} +
+ {more > 0 &&
+{more} more
} +
+ ) +} + +export function CNPGOverview({ + query, + fleet, + namespaces, + searchParams, + onSetParams, + onInspect, + inspected, + onClearNamespaces, +}: { + query: UseQueryResult + fleet: CNPGFleet | null + namespaces: string[] + searchParams: URLSearchParams + onSetParams: (update: Record) => void + onInspect: (resource: SelectedResource) => void + inspected: SelectedResource | null + onClearNamespaces: () => void +}) { + const navigate = useNavigate() + const { connection } = useConnection() + const data = query.data + const q = searchParams.get('q') ?? '' + const cat = (searchParams.get('cat') as CNPGProblemCategory | null) ?? null + const rawFilter = searchParams.get('filter') as Filter | null + const filter: Filter = rawFilter ?? (fleet && fleet.attentionCount > 0 ? 'attention' : 'all') + + const rows = useMemo(() => { + if (!fleet) return [] + let list = fleet.rows + if (filter === 'attention') list = list.filter((r) => r.attention) + if (cat) list = list.filter((r) => r.categories.has(cat)) + if (q) { + const needle = q.toLowerCase() + list = list.filter((r) => r.name.toLowerCase().includes(needle) || r.namespace.toLowerCase().includes(needle)) + } + return list + }, [fleet, filter, cat, q]) + + if (!data && query.isLoading) return + if (!data) { + return ( + + ) + } + if (!data.installed || !fleet) { + return ( + + ) + } + + const clustersCov = data.coverage.clusters + const total = fleet.rows.length + const context = connection.context || data.context + + if (total === 0) { + const denied = clustersCov?.state === 'denied' + return ( +
+ + 0 + ? `None in namespace ${namespaces.join(', ')}. Clear the namespace filter to see the whole cluster.` + : 'The CloudNativePG CRDs are installed. Clusters, backups and declarations appear here once they exist.' + } + action={ + namespaces.length > 0 ? ( + + ) : undefined + } + /> +
+ ) + } + + const chips: { label: string; onClear: () => void }[] = [] + if (cat) chips.push({ label: `Problem: ${CNPG_PROBLEM_CATEGORIES.find((c) => c.id === cat)?.label ?? cat}`, onClear: () => onSetParams({ cat: null }) }) + if (q) chips.push({ label: `Search: ${q}`, onClear: () => onSetParams({ q: null }) }) + if (namespaces.length > 0) chips.push({ label: `Namespace: ${namespaces.join(', ')}`, onClear: onClearNamespaces }) + + const segment = (id: Filter, label: string, n: number) => { + const on = filter === id + return ( + + ) + } + + const lowerBound = fleet.incompleteKinds.length > 0 + + return ( +
+ + {total} PostgreSQL {total === 1 ? 'cluster' : 'clusters'} · {fleet.attentionCount} + {lowerBound ? '+' : ''} need attention + · {context} + + } + /> +
+
+ + +
+
+ {segment('attention', 'Needs attention', fleet.attentionCount)} + {segment('all', 'All clusters', total)} +
+ {CNPG_PROBLEM_CATEGORIES.filter((c) => fleet.categoryCounts[c.id] > 0).map((c) => { + const on = cat === c.id + return ( + + ) + })} +
+ + onSetParams({ q: e.target.value })} + placeholder="Filter clusters…" + aria-label="Filter clusters" + className="min-w-0 flex-1 bg-transparent text-sm text-theme-text-primary placeholder-theme-text-disabled focus:outline-none" + /> +
+
+ + {chips.length > 0 && ( +
+ {chips.map((c) => ( + + {c.label} + + + ))} +
+ )} + +
+
+ + + + + + + + + + + + + + + + + + + + + + + + + {rows.map((row) => { + const ref: SelectedResource = { kind: 'clusters', group: 'postgresql.cnpg.io', namespace: row.namespace, name: row.name } + const active = sameResource(inspected, ref) + return ( + onInspect(ref)} + className={clsx('cursor-pointer', ROW_HOVER, active && 'selection')} + aria-selected={active} + > + + + + + + + + + + ) + })} + +
ClusterReadyReplicationProtectionDeclarationsPGNeeds attentionActions
+
+ p.severity === 'critical') ? 'unhealthy' : 'degraded') : row.controllerStatus.level} /> + {row.name} +
+
{row.namespace}
+
+
+ {row.instances.ready ?? '–'}/{row.instances.desired ?? '–'} + {row.pgVersion ?? '—'} + +
+
+ {rows.length === 0 && ( +
+ {filter === 'attention' && !cat && !q + ? 'No clusters need attention.' + : 'No clusters match these filters.'}{' '} + +
+ )} +
+
+
+
+ ) +} diff --git a/web/src/components/cnpg/CNPGSummaryHost.tsx b/web/src/components/cnpg/CNPGSummaryHost.tsx new file mode 100644 index 000000000..10f9d3b39 --- /dev/null +++ b/web/src/components/cnpg/CNPGSummaryHost.tsx @@ -0,0 +1,52 @@ +import type { ReactNode } from 'react' +import { useNavigate } from 'react-router-dom' +import { CNPGClusterSummary, PaneLoader, refToSelectedResource, type CNPGRef, type NavigateToResource } from '@skyhook-io/k8s-ui' +import { useCNPGFleet } from './useCNPGSidebarWorkspace' +import { cnpgClusterFullPath } from './paths' + +interface SummaryContext { + apiKind: string + namespace: string + name: string + group?: string + resource: any + context: 'drawer' | 'expanded' + onNavigate?: NavigateToResource +} + +function ClusterSummaryHost({ namespace, name, context, onNavigate }: SummaryContext) { + const navigate = useNavigate() + // The workspace is read for the object's own namespace: an explicitly opened + // Cluster shows its facts whatever the namespace filter is. + const { query, fleet } = useCNPGFleet([namespace]) + const row = fleet?.rows.find((r) => r.namespace === namespace && r.name === name) + if (!row) { + if (query.isLoading) return + return ( +
+ {query.error instanceof Error + ? `The CloudNativePG summary could not be loaded: ${query.error.message}` + : 'This Cluster is not in the CloudNativePG workspace for your identity.'}{' '} + Spec & status still shows everything the object reports. +
+ ) + } + const go = onNavigate ? (ref: CNPGRef) => onNavigate(refToSelectedResource(ref)) : undefined + return ( + navigate(cnpgClusterFullPath(namespace, name)) }] : undefined} + /> + ) +} + +/** + * The composed Overview for CloudNativePG kinds. Returns null for kinds + * without one, which keeps the default Overview. + */ +export function renderCNPGSummary(ctx: SummaryContext): ReactNode { + if (ctx.group !== 'postgresql.cnpg.io') return null + if (ctx.resource?.kind === 'Cluster') return + return null +} diff --git a/web/src/components/cnpg/CNPGView.tsx b/web/src/components/cnpg/CNPGView.tsx new file mode 100644 index 000000000..f6ba9c890 --- /dev/null +++ b/web/src/components/cnpg/CNPGView.tsx @@ -0,0 +1,126 @@ +import { useCallback, useEffect, useMemo, useRef } from 'react' +import { useLocation, useNavigate, useSearchParams } from 'react-router-dom' +import { ResourcesSidebar, type SelectedKindInfo } from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { useAPIResources } from '../../api/apiResources' +import { usePinnedKinds } from '../../hooks/useFavorites' +import { useResourceCounts } from '../../hooks/useResourceCounts' +import { CNPGOverview } from './CNPGOverview' +import { decodeDrawerTrail, encodeDrawerTrail, parseCNPGRoute, sameResource } from './routes' +import { useCNPGFleet, useCNPGSidebarWorkspace } from './useCNPGSidebarWorkspace' + +interface CNPGViewProps { + namespaces: string[] + selectedResource: SelectedResource | null + onOpenResource: (resource: SelectedResource) => void + onCloseResource: () => void + onClearNamespaces: () => void +} + +/** + * The CloudNativePG workspace. Lives inside Resources (the rail keeps + * Resources highlighted) and renders the same Resources sidebar, with the + * workspace destinations above the exact CNPG kinds. + * + * The drawer is the app's single resource drawer. `?drawer=` backs it so a + * refresh, a shared link and browser Back restore the inspected object. + */ +export function CNPGView({ namespaces, selectedResource, onOpenResource, onCloseResource, onClearNamespaces }: CNPGViewProps) { + const location = useLocation() + const navigate = useNavigate() + const [searchParams, setSearchParams] = useSearchParams() + const route = parseCNPGRoute(location.pathname) + const { data: apiResources } = useAPIResources() + const { data: counts } = useResourceCounts(namespaces) + const { pinned, togglePin, isPinned } = usePinnedKinds() + const sidebarWorkspace = useCNPGSidebarWorkspace({ apiResources, namespaces, active: { screen: route.screen } }) + const { query, fleet } = useCNPGFleet(namespaces) + + const drawerParam = searchParams.get('drawer') + const trail = useMemo(() => decodeDrawerTrail(drawerParam), [drawerParam]) + const drawerTarget = trail.length > 0 ? trail[trail.length - 1] : null + + // Two-way sync between ?drawer= and the app drawer. Whichever side changed + // since the last sync wins, so URL navigation (Back, a pasted link) opens the + // drawer and drawer navigation (close, a link inside it) rewrites the URL. + const lastSynced = useRef(null) + const selectedKey = selectedResource ? encodeDrawerTrail([selectedResource]) : '' + const targetKey = drawerTarget ? encodeDrawerTrail([drawerTarget]) : '' + useEffect(() => { + if (targetKey !== (lastSynced.current ?? '')) { + lastSynced.current = targetKey + if (drawerTarget && !sameResource(drawerTarget, selectedResource)) onOpenResource(drawerTarget) + else if (!drawerTarget && selectedResource) onCloseResource() + return + } + if (selectedKey !== targetKey) { + lastSynced.current = selectedKey + const params = new URLSearchParams(searchParams) + if (!selectedResource) { + params.delete('drawer') + } else { + const idx = trail.findIndex((r) => sameResource(r, selectedResource)) + const next = idx >= 0 ? trail.slice(0, idx + 1) : [...trail, selectedResource] + params.set('drawer', encodeDrawerTrail(next)) + } + setSearchParams(params, { replace: true }) + } + }, [targetKey, selectedKey]) // eslint-disable-line react-hooks/exhaustive-deps + + const inspect = useCallback( + (resource: SelectedResource) => { + const params = new URLSearchParams(searchParams) + params.set('drawer', encodeDrawerTrail([resource])) + setSearchParams(params, { replace: true }) + }, + [searchParams, setSearchParams], + ) + + const setParams = useCallback( + (update: Record) => { + const params = new URLSearchParams(searchParams) + for (const [k, v] of Object.entries(update)) { + if (v === null || v === '') params.delete(k) + else params.set(k, v) + } + setSearchParams(params, { replace: true }) + }, + [searchParams, setSearchParams], + ) + + const selectKind = useCallback( + (kind: SelectedKindInfo) => { + navigate(`/resources/${kind.name}${kind.group ? `?apiGroup=${encodeURIComponent(kind.group)}` : ''}`) + }, + [navigate], + ) + + return ( +
+ isPinned(kind, group ?? '')} + categoryWorkspaces={sidebarWorkspace} + /> +
+ +
+
+ ) +} diff --git a/web/src/components/cnpg/paths.ts b/web/src/components/cnpg/paths.ts new file mode 100644 index 000000000..e42b6571d --- /dev/null +++ b/web/src/components/cnpg/paths.ts @@ -0,0 +1,5 @@ +import { buildWorkloadPath } from '../../utils/navigation' + +export function cnpgClusterFullPath(namespace: string, name: string): string { + return buildWorkloadPath({ kind: 'clusters', namespace, name, group: 'postgresql.cnpg.io' }) +} diff --git a/web/src/components/cnpg/routes.test.ts b/web/src/components/cnpg/routes.test.ts new file mode 100644 index 000000000..fb19f5209 --- /dev/null +++ b/web/src/components/cnpg/routes.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from 'vitest' +import { decodeDrawerTrail, encodeDrawerTrail, parseCNPGRoute, sameResource } from './routes' + +describe('CNPG routes', () => { + it('parses workspace screens and falls back to Overview for unknown or unavailable ones', () => { + expect(parseCNPGRoute('/cnpg').screen).toBe('overview') + expect(parseCNPGRoute('/cnpg/').screen).toBe('overview') + expect(parseCNPGRoute('/cnpg/nope').screen).toBe('overview') + }) + + it('round-trips a drawer trail and keeps the API group', () => { + const trail = [ + { kind: 'backups', group: 'postgresql.cnpg.io', namespace: 'payments', name: 'pg-billing-20260928020000' }, + { kind: 'objectstores', group: 'barmancloud.cnpg.io', namespace: 'payments', name: 's3-billing' }, + { kind: 'clusterimagecatalogs', group: 'postgresql.cnpg.io', namespace: '', name: 'postgresql-standard' }, + ] + const encoded = encodeDrawerTrail(trail) + expect(decodeDrawerTrail(encoded)).toEqual(trail) + }) + + it('drops malformed entries instead of opening a guessed object', () => { + expect(decodeDrawerTrail('clusters:postgresql.cnpg.io:payments:pg-orders~garbage')).toEqual([ + { kind: 'clusters', group: 'postgresql.cnpg.io', namespace: 'payments', name: 'pg-orders' }, + ]) + expect(decodeDrawerTrail(null)).toEqual([]) + }) + + it('distinguishes same-named kinds from different groups', () => { + const cnpg = { kind: 'clusters', group: 'postgresql.cnpg.io', namespace: 'a', name: 'x' } + const capi = { kind: 'clusters', group: 'cluster.x-k8s.io', namespace: 'a', name: 'x' } + expect(sameResource(cnpg, capi)).toBe(false) + expect(sameResource(cnpg, { ...cnpg })).toBe(true) + }) +}) diff --git a/web/src/components/cnpg/routes.ts b/web/src/components/cnpg/routes.ts new file mode 100644 index 000000000..e4d4b609a --- /dev/null +++ b/web/src/components/cnpg/routes.ts @@ -0,0 +1,64 @@ +import type { SelectedResource } from '../../types' + +export type CNPGScreen = 'overview' | 'protection' | 'declarations' | 'pooling' | 'operator' + +export const CNPG_SCREENS: { id: CNPGScreen; label: string; path: string }[] = [ + { id: 'overview', label: 'Overview', path: '/cnpg' }, + { id: 'protection', label: 'Protection', path: '/cnpg/protection' }, + { id: 'declarations', label: 'Declarations', path: '/cnpg/declarations' }, + { id: 'pooling', label: 'Pooling', path: '/cnpg/pooling' }, + { id: 'operator', label: 'Operator', path: '/cnpg/operator' }, +] + +/** Screens that exist in this build. Destinations appear in the sidebar only once their screen does. */ +export const CNPG_AVAILABLE_SCREENS: ReadonlySet = new Set(['overview']) + +export interface CNPGRoute { + screen: CNPGScreen +} + +export function parseCNPGRoute(pathname: string): CNPGRoute { + const seg = pathname.replace(/^\/+/, '').split('/') + if (seg[0] !== 'cnpg') return { screen: 'overview' } + const s = seg[1] ?? '' + const match = CNPG_SCREENS.find((x) => x.id === s) + if (match && CNPG_AVAILABLE_SCREENS.has(match.id)) return { screen: match.id } + return { screen: 'overview' } +} + +export function cnpgScreenPath(screen: CNPGScreen): string { + return CNPG_SCREENS.find((s) => s.id === screen)?.path ?? '/cnpg' +} + +// Drawer identity in the URL: kind:group:namespace:name, chained with "~" for +// the in-drawer trail (last entry is the one shown). Kubernetes names and API +// groups cannot contain ":" or "~", and the group is mandatory — CNPG's Cluster +// and Backup collide with CAPI, KubeBlocks and Velero kinds. +export function encodeDrawerRef(r: SelectedResource): string { + return [r.kind, r.group ?? '', r.namespace ?? '', r.name].join(':') +} + +export function decodeDrawerRef(s: string): SelectedResource | null { + const parts = s.split(':') + if (parts.length !== 4 || !parts[0] || !parts[3]) return null + return { kind: parts[0], group: parts[1], namespace: parts[2], name: parts[3] } +} + +export function decodeDrawerTrail(param: string | null): SelectedResource[] { + if (!param) return [] + return param.split('~').map(decodeDrawerRef).filter((r): r is SelectedResource => r !== null) +} + +export function encodeDrawerTrail(trail: SelectedResource[]): string { + return trail.map(encodeDrawerRef).join('~') +} + +export function sameResource(a: SelectedResource | null | undefined, b: SelectedResource | null | undefined): boolean { + if (!a || !b) return false + return ( + a.kind.toLowerCase() === b.kind.toLowerCase() && + (a.group ?? '') === (b.group ?? '') && + (a.namespace ?? '') === (b.namespace ?? '') && + a.name === b.name + ) +} diff --git a/web/src/components/cnpg/useCNPGSidebarWorkspace.ts b/web/src/components/cnpg/useCNPGSidebarWorkspace.ts new file mode 100644 index 000000000..fbf7568fd --- /dev/null +++ b/web/src/components/cnpg/useCNPGSidebarWorkspace.ts @@ -0,0 +1,88 @@ +import { useMemo } from 'react' +import { useNavigate } from 'react-router-dom' +import { Database, FileCheck2, Settings2, ShieldCheck, Waypoints } from 'lucide-react' +import { buildCNPGFleet, type CNPGFleet, type SidebarCategoryWorkspace } from '@skyhook-io/k8s-ui' +import type { APIResource } from '../../types' +import { useCNPGWorkspace } from '../../api/cnpg' +import { CNPG_AVAILABLE_SCREENS, CNPG_SCREENS, type CNPGScreen } from './routes' + +export const CNPG_SIDEBAR_CATEGORY = 'CloudNativePG' + +const ICONS: Record = { + overview: Database, + protection: ShieldCheck, + declarations: FileCheck2, + pooling: Waypoints, + operator: Settings2, +} + +export function cnpgDiscovered(apiResources: APIResource[] | undefined): boolean { + return !!apiResources?.some((r) => r.group === 'postgresql.cnpg.io') +} + +export function useCNPGFleet(namespaces: string[], enabled = true) { + const query = useCNPGWorkspace(namespaces, { enabled }) + const fleet = useMemo(() => (query.data?.installed ? buildCNPGFleet(query.data) : null), [query.data]) + return { query, fleet } +} + +function destinationCount(screen: CNPGScreen, fleet: CNPGFleet | null): { count?: number | null; title?: string } { + if (!fleet) return {} + const lowerBound = fleet.incompleteKinds.length > 0 ? ' Some CloudNativePG data is not readable, so this is a lower bound.' : '' + switch (screen) { + case 'overview': + return { count: fleet.attentionCount, title: `${fleet.attentionCount} clusters need attention.${lowerBound}` } + case 'protection': + return { count: fleet.categoryCounts.protection, title: `${fleet.categoryCounts.protection} clusters with failing backups or WAL archiving.${lowerBound}` } + case 'declarations': + return { count: fleet.categoryCounts.declarations, title: `${fleet.categoryCounts.declarations} clusters with declarations that are not reconciled.${lowerBound}` } + case 'pooling': + return { count: fleet.categoryCounts.pooling, title: `${fleet.categoryCounts.pooling} clusters with Pooler problems.${lowerBound}` } + default: + return {} + } +} + +/** + * The CloudNativePG workspace entries for the Resources sidebar. Returns + * undefined when CNPG is not discovered, so clusters without the operator see + * an unchanged sidebar. + */ +export function useCNPGSidebarWorkspace({ + apiResources, + namespaces, + active, +}: { + apiResources: APIResource[] | undefined + namespaces: string[] + active?: { screen: CNPGScreen; child?: { label: string; title?: string } } +}): Record | undefined { + const navigate = useNavigate() + const discovered = cnpgDiscovered(apiResources) + const { fleet } = useCNPGFleet(namespaces, discovered) + const nsKey = namespaces.join(',') + + return useMemo(() => { + if (!discovered) return undefined + const destinations = CNPG_SCREENS.filter((s) => CNPG_AVAILABLE_SCREENS.has(s.id)).map((s) => { + const { count, title } = destinationCount(s.id, fleet) + return { + id: s.id, + label: s.label, + icon: ICONS[s.id], + count, + countTitle: title, + active: active?.screen === s.id, + child: active?.screen === s.id ? active.child : undefined, + onSelect: () => navigate(s.path), + } + }) + return { + [CNPG_SIDEBAR_CATEGORY]: { + destinations, + defaultKindsCollapsed: !!active, + scopeNote: nsKey ? `Counts for namespace ${nsKey.split(',').join(', ')}` : undefined, + }, + } + }, [discovered, fleet, active?.screen, active?.child?.label, active?.child?.title, navigate, nsKey]) // eslint-disable-line react-hooks/exhaustive-deps +} diff --git a/web/src/components/resources/ResourcesView.tsx b/web/src/components/resources/ResourcesView.tsx index 2aa61b915..b4d1ee9d2 100644 --- a/web/src/components/resources/ResourcesView.tsx +++ b/web/src/components/resources/ResourcesView.tsx @@ -9,6 +9,8 @@ import { useAPIResources } from '../../api/apiResources' import { useConnection } from '../../context/ConnectionContext' import { initNavigationMap, getSecretStoreProviderType } from '@skyhook-io/k8s-ui' import { usePinnedKinds } from '../../hooks/useFavorites' +import { useResourceCounts } from '../../hooks/useResourceCounts' +import { useCNPGSidebarWorkspace } from '../cnpg/useCNPGSidebarWorkspace' import { useOpenLogs, useOpenWorkloadLogs } from '../dock' import { canBulkRestartKind, @@ -25,13 +27,6 @@ import { apiVersionToGroup, kindToPluralWithGroup, type NavigateToResource } fro import { CreateResourceDialog } from '../shared/CreateResourceDialog' import { getSkeletonYaml } from '../../utils/skeleton-yaml' -interface ResourceCountsResponse { - counts: Record - forbidden?: string[] - reasons?: Record - unavailable?: string[] -} - interface ResourcesViewProps { namespaces: string[] selectedResource?: SelectedResource | null @@ -113,6 +108,8 @@ export function ResourcesView({ namespaces, selectedResource, onResourceClick, o if (apiResources) initNavigationMap(apiResources) }, [apiResources]) + const cnpgSidebarWorkspace = useCNPGSidebarWorkspace({ apiResources, namespaces }) + // Track the selected kind from the k8s-ui component const [selectedKind, setSelectedKind] = useState(null) const workloadWrites = namespaces.length === 0 @@ -125,37 +122,7 @@ export function ResourcesView({ namespaces, selectedResource, onResourceClick, o // Lightweight resource counts for sidebar badges (~2KB instead of ~608MB) const namespacesParam = namespaces.join(',') - const { data: countsData, isError: countsIsError } = useQuery({ - queryKey: ['resource-counts', namespacesParam], - queryFn: async () => { - const params = new URLSearchParams() - if (namespaces.length > 0) params.set('namespaces', namespacesParam) - const startedAt = performance.now() - debugNamespaceLog('resources:counts-fetch-start', { namespaces, params: params.toString() }) - try { - return await fetchJSON(`/resource-counts?${params}`) - } finally { - debugNamespaceLog('resources:counts-fetch-end', { - namespaces, - params: params.toString(), - durationMs: Math.round(performance.now() - startedAt), - }) - } - }, - staleTime: 10000, - // SSE invalidation isn't running while connecting, and mid-sync counts - // are what unlatch guarded kinds as their informers finish — poll fast - // during the shell, settle to the safety net once connected. - refetchInterval: connection.state === 'connecting' ? 3000 : 60000, - // During the first seconds of the progressive shell the endpoint 503s - // (cluster_connecting) until the mid-sync cache handle exists; keep the - // query pending rather than parking it in error state, which would - // unlatch the large-list guard at the connected flip. - retry: (failureCount: number, error: Error) => - isStillLoadingError(error) ? true : failureCount < 3, - retryDelay: (failureCount: number, error: Error) => - isStillLoadingError(error) ? 2000 : Math.min(1000 * 2 ** failureCount, 30000), - }) + const { data: countsData, isError: countsIsError } = useResourceCounts(namespaces) // Determine if selected kind is a CRD (only CRDs should send ?group= to backend) const isSelectedCrd = useMemo(() => { @@ -439,6 +406,7 @@ export function ResourcesView({ namespaces, selectedResource, onResourceClick, o connectionState={connection.state === 'connecting' && connection.syncStatus ? 'syncing' : connection.state} largeListGuard={largeListGuard} onSelectedKindChange={setSelectedKind} + sidebarCategoryWorkspaces={cnpgSidebarWorkspace} topPodMetrics={topPodMetrics} topNodeMetrics={topNodeMetrics} certExpiry={certExpiry} diff --git a/web/src/components/workload/WorkloadView.tsx b/web/src/components/workload/WorkloadView.tsx index e4f1a5c89..81670bc37 100644 --- a/web/src/components/workload/WorkloadView.tsx +++ b/web/src/components/workload/WorkloadView.tsx @@ -147,6 +147,7 @@ import { CNPGSubscriptionRenderer, } from '../resources/renderers/CNPGDeclarativeRenderer' import { CreateResourceDialog } from '../shared/CreateResourceDialog' +import { renderCNPGSummary } from '../cnpg/CNPGSummaryHost' import { cleanYamlForDuplicate } from '../../utils/skeleton-yaml' import { useDesktopDownload } from '../../hooks/useDesktopDownload' import { useCompareLauncher } from '../compare/useCompareLauncher' @@ -1247,6 +1248,9 @@ export function WorkloadView({ onSelectRun={handleSelectedRunChange} /> )} + renderSummary={({ apiKind: ak, namespace: ns, name: n, resource: res, context, onNavigate }) => + renderCNPGSummary({ apiKind: ak, namespace: ns, name: n, group: effectiveGroup, resource: res, context, onNavigate }) + } renderExpandedOverview={({ kind: k, apiKind, namespace: ns, name: n, resource: res }) => supportsBatchExecution(k, apiKind, effectiveGroup, res?.apiVersion) && res ? ( diff --git a/web/src/hooks/useResourceCounts.ts b/web/src/hooks/useResourceCounts.ts new file mode 100644 index 000000000..c1b1396fb --- /dev/null +++ b/web/src/hooks/useResourceCounts.ts @@ -0,0 +1,49 @@ +import { useQuery } from '@tanstack/react-query' +import { debugNamespaceLog, fetchJSON, isStillLoadingError } from '../api/client' +import { useConnection } from '../context/ConnectionContext' + +export interface ResourceCountsResponse { + counts: Record + forbidden?: string[] + reasons?: Record + unavailable?: string[] +} + +// Lightweight per-kind counts for the Resources sidebar badges. Shared by the +// Resources view and any surface that renders the same sidebar standalone, so +// both read one cache entry. +export function useResourceCounts(namespaces: string[]) { + const { connection } = useConnection() + const namespacesParam = namespaces.join(',') + return useQuery({ + queryKey: ['resource-counts', namespacesParam], + queryFn: async () => { + const params = new URLSearchParams() + if (namespaces.length > 0) params.set('namespaces', namespacesParam) + const startedAt = performance.now() + debugNamespaceLog('resources:counts-fetch-start', { namespaces, params: params.toString() }) + try { + return await fetchJSON(`/resource-counts?${params}`) + } finally { + debugNamespaceLog('resources:counts-fetch-end', { + namespaces, + params: params.toString(), + durationMs: Math.round(performance.now() - startedAt), + }) + } + }, + staleTime: 10000, + // SSE invalidation isn't running while connecting, and mid-sync counts + // are what unlatch guarded kinds as their informers finish — poll fast + // during the shell, settle to the safety net once connected. + refetchInterval: connection.state === 'connecting' ? 3000 : 60000, + // During the first seconds of the progressive shell the endpoint 503s + // (cluster_connecting) until the mid-sync cache handle exists; keep the + // query pending rather than parking it in error state, which would + // unlatch the large-list guard at the connected flip. + retry: (failureCount: number, error: Error) => + isStillLoadingError(error) ? true : failureCount < 3, + retryDelay: (failureCount: number, error: Error) => + isStillLoadingError(error) ? 2000 : Math.min(1000 * 2 ** failureCount, 30000), + }) +} From 91241a9135343514d717a1b21bf38cdbfcec73d0 Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Tue, 29 Sep 2026 02:26:17 +0300 Subject: [PATCH 02/17] Add CloudNativePG Protection, Declarations, Pooling and Operator screens and composed summaries Screens (/cnpg/protection, /declarations, /pooling, /operator): - Protection keeps schedule, destination, last successful backup (with its source), WAL archiving, recovery window and restore validation as separate facts; ObjectStore upload health is labelled inferred with its evidence. - Declarations groups Databases, Publications, Subscriptions and managed roles by cluster, separating applied, not applied and pending, with controller errors and GitOps source. - Pooling lists Poolers with readiness; connection pressure reads "Not measured" until PgBouncer metrics are read. - Operator shows operator/plugin workloads and versions, image catalogs and their visible users, and operator configuration (Secret contents never read). Composed Overview summaries (Spec & status keeps the existing renderers) for Backup, ScheduledBackup, ObjectStore, Database, Publication, Subscription, Pooler and image catalogs; drawer trail back-link on workspace screens. GET /api/cnpg/operator discovers operator and plugin Deployments and their config references, gated per namespace on list deployments/services and get configmaps; it does not follow the namespace view filter. Workspace hardening from review: - issues are read flat, so Pod evidence only reaches callers with Pod access; - instance Pods must be controller-owned by the visible Cluster's UID; - partial coverage names denied namespaces only when the caller supplied the candidates, and always reports the allowed ones; - unobserved managed roles are pending, unreadable schedules or Poolers are unknown rather than absent, restore evidence requires a ready restored cluster resolved by exact reference, and an unreadable Cluster list is not reported as "no clusters". --- CLAUDE.md | 1 + internal/server/cnpg_operator.go | 366 ++++++++++++++++ internal/server/cnpg_operator_test.go | 368 ++++++++++++++++ internal/server/cnpg_workspace.go | 176 +++++--- internal/server/cnpg_workspace_test.go | 138 +++++- internal/server/server.go | 1 + .../src/components/cnpg/CNPGBackupSummary.tsx | 211 ++++++++++ .../components/cnpg/CNPGClusterSummary.tsx | 4 +- .../cnpg/CNPGDeclarativeSummary.tsx | 264 ++++++++++++ .../cnpg/CNPGImageCatalogSummary.tsx | 76 ++++ .../cnpg/CNPGObjectStoreSummary.tsx | 174 ++++++++ .../cnpg/CNPGObjectSummary.test.tsx | 193 +++++++++ .../src/components/cnpg/CNPGPoolerSummary.tsx | 70 +++ .../src/components/cnpg/CNPGSharedSummary.tsx | 88 ++++ packages/k8s-ui/src/components/cnpg/index.ts | 5 + .../k8s-ui/src/components/cnpg/primitives.tsx | 5 +- .../src/components/cnpg/relations.test.ts | 289 +++++++++++++ .../k8s-ui/src/components/cnpg/relations.ts | 398 ++++++++++++++++++ .../src/components/cnpg/workspace.test.ts | 54 +++ .../k8s-ui/src/components/cnpg/workspace.ts | 89 ++-- web/src/api/cnpg.ts | 48 +++ web/src/components/cnpg/CNPGDeclarations.tsx | 303 +++++++++++++ web/src/components/cnpg/CNPGOperator.tsx | 226 ++++++++++ web/src/components/cnpg/CNPGOverview.tsx | 131 +----- web/src/components/cnpg/CNPGPooling.tsx | 116 +++++ web/src/components/cnpg/CNPGProtection.tsx | 299 +++++++++++++ web/src/components/cnpg/CNPGSummaryHost.tsx | 108 ++++- web/src/components/cnpg/CNPGView.tsx | 41 +- web/src/components/cnpg/routes.test.ts | 2 + web/src/components/cnpg/routes.ts | 5 +- web/src/components/cnpg/shared.tsx | 265 ++++++++++++ .../cnpg/useCNPGSidebarWorkspace.ts | 4 +- 32 files changed, 4294 insertions(+), 224 deletions(-) create mode 100644 internal/server/cnpg_operator.go create mode 100644 internal/server/cnpg_operator_test.go create mode 100644 packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx create mode 100644 packages/k8s-ui/src/components/cnpg/relations.test.ts create mode 100644 packages/k8s-ui/src/components/cnpg/relations.ts create mode 100644 web/src/components/cnpg/CNPGDeclarations.tsx create mode 100644 web/src/components/cnpg/CNPGOperator.tsx create mode 100644 web/src/components/cnpg/CNPGPooling.tsx create mode 100644 web/src/components/cnpg/CNPGProtection.tsx create mode 100644 web/src/components/cnpg/shared.tsx diff --git a/CLAUDE.md b/CLAUDE.md index 280cb2ed4..48d40150b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -183,6 +183,7 @@ After `make -demo`, run `kubectl config use-context kind-radar--demo - Policy (Kyverno) reverse-lookup: `/api/policy/resource/{kind}/{ns}/{name}` returns one resource's policy findings; `/api/policy/policies/{policy}?namespace=&limit=` returns the inverse — every resource one policy recorded an outcome for, per rule. Report families are authorized **per subject scope** (`policyreports` cluster-wide is a different grant from `clusterpolicyreports`), findings from an unreadable family are dropped from lists AND counts with the withheld count reported, and `counts` describe the cluster while subject lists are capped and follow the namespace view filter — the two must be read together. `/api/policy/policies/{policy}/queued` returns the policy's in-flight `UpdateRequest`s; Kyverno records those in its own namespace, so it reads cluster-wide gated on `list updaterequests` rather than inheriting the caller's view filter. - Velero reverse-lookup: `/api/velero/backupstoragelocations/{ns}/{name}/backups` returns the Backups a storage location holds, with each one's phase, completion time and expiration, plus how many reached `Completed`. Gated on `list backups`; an unset `spec.storageLocation` resolves to whichever location carries `spec.default`, falling back to the name `default` when none does. Backs the storage location's "Stored Here" section, which states what an `Unavailable` location is holding back — the Backup's own status cannot see its location's health. `POST /api/velero/{backups|restores}/{ns}/{name}/messages` returns the warnings and errors behind a run's counts: it creates a `DownloadRequest`, waits for Velero to answer with a pre-signed URL, and fetches the results file from object storage. Impersonated (it creates a CR); needs a running Velero controller and object storage reachable from wherever Radar runs, and reports which of the two failed rather than returning an empty list. - CloudNativePG workspace: `/api/cnpg/workspace` returns every CNPG kind (plus owner-validated instance Pods) with per-kind `coverage` (`full|partial|denied|notInstalled|syncing|error`, `partial` naming only in-scope denied namespaces), CNPG issues from the Issues engine and `cnpgNoDeclarativeBackup` audit findings, each withheld where the caller lacks coverage. Namespaced kinds follow the view filter and the capacity per-namespace `list` fallback; `ClusterImageCatalog` needs a cluster-scope `list`. Handler `internal/server/cnpg_workspace.go` +- CloudNativePG operator: `/api/cnpg/operator` returns the operator Deployments (`app.kubernetes.io/name=cloudnative-pg`) and plugin Deployments (served by Services labelled `cnpg.io/pluginName`) with image-tag version and readiness (`null` when unreported), plus the operator's ConfigMap/Secret/monitoring-queries references from its args and env. ConfigMap data only with `get configmaps`; the Secret is name-only, never read. Deployments and Services carry the workspace `coverage` states; deliberately ignores the namespace view filter (the operator lives in its own namespace). Handler `internal/server/cnpg_operator.go` - CloudNativePG reverse-lookup: `/api/cnpg/imagecatalogs/{ns}/{name}/clusters` and `/api/cnpg/clusterimagecatalogs/{name}/clusters` return the Clusters pinned to an image catalog, with the major each asks for and the image it actually resolved. Cluster-scoped catalogs are referenceable from any namespace, so the cluster-scoped route reads cluster-wide gated on `list clusters` — a view-filtered answer would report "nothing uses this" before an edit. ## Key Patterns diff --git a/internal/server/cnpg_operator.go b/internal/server/cnpg_operator.go new file mode 100644 index 000000000..2d4871675 --- /dev/null +++ b/internal/server/cnpg_operator.go @@ -0,0 +1,366 @@ +package server + +import ( + "log" + "net/http" + "sort" + "strings" + + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/labels" + + "github.com/skyhook-io/radar/internal/k8s" +) + +const ( + cnpgOperatorNameLabel = "app.kubernetes.io/name" + cnpgOperatorNameValue = "cloudnative-pg" + cnpgVersionLabel = "app.kubernetes.io/version" + cnpgPluginNameLabel = "cnpg.io/pluginName" + cnpgOperatorContainer = "manager" + cnpgOperatorDeployVar = "OPERATOR_DEPLOYMENT_NAME" + cnpgMonitoringQueriesCM = "MONITORING_QUERIES_CONFIGMAP" + + cnpgOperatorRoleOperator = "operator" + cnpgOperatorRolePlugin = "plugin" + + cnpgConfigPurposeOperator = "operator" + cnpgConfigPurposeMonitoring = "monitoring" +) + +// CNPGOperatorComponent is one operator or plugin Deployment. Version is the +// image tag, else the app.kubernetes.io/version label, else empty. Replica +// counts are nil when unreported, which is not zero. +type CNPGOperatorComponent struct { + Role string `json:"role"` + PluginName string `json:"pluginName,omitempty"` + Namespace string `json:"namespace"` + Deployment string `json:"deployment"` + Image string `json:"image"` + Version string `json:"version"` + ReadyReplicas *int32 `json:"readyReplicas"` + Replicas *int32 `json:"replicas"` +} + +// CNPGOperatorConfigMapState is present only on ConfigMap references. A Secret +// reference never carries it: the endpoint never reads Secrets. +type CNPGOperatorConfigMapState struct { + Exists *bool `json:"exists"` + Readable bool `json:"readable"` + Reason string `json:"reason,omitempty"` + Data map[string]string `json:"data"` +} + +// CNPGOperatorConfigRef is a ConfigMap or Secret the operator is configured +// to read. +type CNPGOperatorConfigRef struct { + Kind string `json:"kind"` + Namespace string `json:"namespace"` + Name string `json:"name"` + Purpose string `json:"purpose"` + *CNPGOperatorConfigMapState +} + +// CNPGOperatorResponse is GET /api/cnpg/operator. +type CNPGOperatorResponse struct { + Coverage map[string]CNPGWorkspaceCoverage `json:"coverage"` + Components []CNPGOperatorComponent `json:"components"` + Config []CNPGOperatorConfigRef `json:"config"` +} + +// handleCNPGOperator serves GET /api/cnpg/operator: the operator and plugin +// Deployments, their versions and readiness, and where the operator's +// configuration lives. +// +// The operator runs in its own namespace (cnpg-system by default) while +// people filter the view to their application namespaces. Following the view +// filter would report "no operator" to anyone looking at their databases, so +// scope follows permission here, as it does for the catalog reverse lookups. +func (s *Server) handleCNPGOperator(w http.ResponseWriter, r *http.Request) { + if !s.requireConnected(w) { + return + } + cache := k8s.GetResourceCache() + if cache == nil { + s.writeError(w, http.StatusServiceUnavailable, "Resource cache not available") + return + } + + scope := s.cnpgOperatorScope(r) + resp := CNPGOperatorResponse{ + Coverage: map[string]CNPGWorkspaceCoverage{}, + Components: []CNPGOperatorComponent{}, + Config: []CNPGOperatorConfigRef{}, + } + + depAcc, depDenied, deployments := s.cnpgOperatorDeployments(r, cache, scope) + resp.Coverage["deployments"] = cnpgCoverageOf(depAcc, depDenied) + svcAcc, svcDenied, services := s.cnpgOperatorServices(r, cache, scope) + resp.Coverage["services"] = cnpgCoverageOf(svcAcc, svcDenied) + + var operators []*appsv1.Deployment + for _, d := range deployments { + if d.Labels[cnpgOperatorNameLabel] == cnpgOperatorNameValue { + operators = append(operators, d) + resp.Components = append(resp.Components, cnpgOperatorComponent(d, cnpgOperatorRoleOperator, "", cnpgOperatorContainerOf(d))) + } + } + + byNamespace := map[string][]*appsv1.Deployment{} + for _, d := range deployments { + byNamespace[d.Namespace] = append(byNamespace[d.Namespace], d) + } + var plugins []CNPGOperatorComponent + for _, svc := range services { + pluginName := svc.Labels[cnpgPluginNameLabel] + if pluginName == "" || !depAcc.covers(svc.Namespace) { + continue + } + matched := false + if len(svc.Spec.Selector) > 0 { + sel := labels.SelectorFromSet(svc.Spec.Selector) + for _, d := range byNamespace[svc.Namespace] { + if sel.Matches(labels.Set(d.Spec.Template.Labels)) { + matched = true + plugins = append(plugins, cnpgOperatorComponent(d, cnpgOperatorRolePlugin, pluginName, firstContainer(d))) + } + } + } + if !matched { + plugins = append(plugins, CNPGOperatorComponent{Role: cnpgOperatorRolePlugin, PluginName: pluginName, Namespace: svc.Namespace}) + } + } + sort.SliceStable(plugins, func(i, j int) bool { + a, b := plugins[i], plugins[j] + if a.PluginName != b.PluginName { + return a.PluginName < b.PluginName + } + if a.Namespace != b.Namespace { + return a.Namespace < b.Namespace + } + return a.Deployment < b.Deployment + }) + resp.Components = append(resp.Components, plugins...) + + resp.Config = s.cnpgOperatorConfig(r, cache, operators) + s.writeJSON(w, resp) +} + +// cnpgOperatorScope is the caller's RBAC scope without the view filter. A +// Radar forced into one namespace still answers only for that namespace. +func (s *Server) cnpgOperatorScope(r *http.Request) []string { + if k8s.ForceNamespaceScope { + target := k8s.GetNamespaceScopeTarget() + if target == "" { + return []string{} + } + return s.getUserNamespaces(r, []string{target}) + } + return s.getUserNamespaces(r, nil) +} + +func (s *Server) cnpgOperatorDeployments(r *http.Request, cache *k8s.ResourceCache, scope []string) (cnpgKindAccess, []string, []*appsv1.Deployment) { + acc, denied, read := s.cnpgTypedScope(r, cache, scope, "apps", "deployments") + if acc.state == cnpgCoverageDenied || acc.state == cnpgCoverageError { + return acc, denied, nil + } + lister := cache.Deployments() + if lister == nil || !cache.IsKindReady("deployments") { + return cnpgKindAccess{state: cnpgCoverageSyncing}, nil, nil + } + var out []*appsv1.Deployment + if read == nil { + out, _ = lister.List(labels.Everything()) + } else { + for _, ns := range read { + items, _ := lister.Deployments(ns).List(labels.Everything()) + out = append(out, items...) + } + } + sort.Slice(out, func(i, j int) bool { + if out[i].Namespace != out[j].Namespace { + return out[i].Namespace < out[j].Namespace + } + return out[i].Name < out[j].Name + }) + return acc, denied, out +} + +func (s *Server) cnpgOperatorServices(r *http.Request, cache *k8s.ResourceCache, scope []string) (cnpgKindAccess, []string, []*corev1.Service) { + acc, denied, read := s.cnpgTypedScope(r, cache, scope, "", "services") + if acc.state == cnpgCoverageDenied || acc.state == cnpgCoverageError { + return acc, denied, nil + } + lister := cache.Services() + if lister == nil || !cache.IsKindReady("services") { + return cnpgKindAccess{state: cnpgCoverageSyncing}, nil, nil + } + hasPlugin, err := labels.Parse(cnpgPluginNameLabel) + if err != nil { + log.Printf("[cnpg] Failed to build plugin selector: %v", err) + return cnpgKindAccess{state: cnpgCoverageError}, nil, nil + } + var out []*corev1.Service + if read == nil { + out, _ = lister.List(hasPlugin) + } else { + for _, ns := range read { + items, _ := lister.Services(ns).List(hasPlugin) + out = append(out, items...) + } + } + return acc, denied, out +} + +func cnpgOperatorContainerOf(d *appsv1.Deployment) *corev1.Container { + for i := range d.Spec.Template.Spec.Containers { + if d.Spec.Template.Spec.Containers[i].Name == cnpgOperatorContainer { + return &d.Spec.Template.Spec.Containers[i] + } + } + return firstContainer(d) +} + +func firstContainer(d *appsv1.Deployment) *corev1.Container { + if len(d.Spec.Template.Spec.Containers) == 0 { + return nil + } + return &d.Spec.Template.Spec.Containers[0] +} + +func cnpgOperatorComponent(d *appsv1.Deployment, role, pluginName string, c *corev1.Container) CNPGOperatorComponent { + out := CNPGOperatorComponent{ + Role: role, + PluginName: pluginName, + Namespace: d.Namespace, + Deployment: d.Name, + Replicas: d.Spec.Replicas, + } + if c != nil { + out.Image = c.Image + out.Version = imageTag(c.Image) + } + if out.Version == "" { + out.Version = d.Labels[cnpgVersionLabel] + } + if out.Version == "" { + out.Version = d.Spec.Template.Labels[cnpgVersionLabel] + } + // The typed status cannot tell an omitted readyReplicas from zero; a status + // the controller has observed at least once states it authoritatively. + if d.Status.ObservedGeneration > 0 { + ready := d.Status.ReadyReplicas + out.ReadyReplicas = &ready + } + return out +} + +// cnpgOperatorArg returns the value of --flag=value or --flag value from a +// container's command and args. +func cnpgOperatorArg(c *corev1.Container, flag string) string { + argv := append(append([]string{}, c.Command...), c.Args...) + for i, a := range argv { + if v, ok := strings.CutPrefix(a, flag+"="); ok { + return v + } + if a == flag && i+1 < len(argv) { + return argv[i+1] + } + } + return "" +} + +func cnpgOperatorEnv(c *corev1.Container, name string) string { + for _, e := range c.Env { + if e.Name == name && e.ValueFrom == nil { + return e.Value + } + } + return "" +} + +// cnpgOperatorExpand resolves $(OPERATOR_DEPLOYMENT_NAME) the way the kubelet +// would: from the container's literal env, which the shipped manifests set to +// the Deployment's own name. Any other reference is left verbatim, as the +// kubelet leaves an unresolvable one. +func cnpgOperatorExpand(v string, c *corev1.Container, d *appsv1.Deployment) string { + ref := "$(" + cnpgOperatorDeployVar + ")" + if !strings.Contains(v, ref) { + return v + } + name := cnpgOperatorEnv(c, cnpgOperatorDeployVar) + if name == "" { + name = d.Name + } + return strings.ReplaceAll(v, ref, name) +} + +func (s *Server) cnpgOperatorConfig(r *http.Request, cache *k8s.ResourceCache, operators []*appsv1.Deployment) []CNPGOperatorConfigRef { + out := []CNPGOperatorConfigRef{} + seen := map[string]bool{} + add := func(ref CNPGOperatorConfigRef) { + key := ref.Kind + "\x00" + ref.Namespace + "\x00" + ref.Name + "\x00" + ref.Purpose + if ref.Name == "" || seen[key] { + return + } + seen[key] = true + out = append(out, ref) + } + for _, d := range operators { + c := cnpgOperatorContainerOf(d) + if c == nil { + continue + } + if name := cnpgOperatorExpand(cnpgOperatorArg(c, "--config-map-name"), c, d); name != "" { + add(s.cnpgOperatorConfigMap(r, cache, d.Namespace, name, cnpgConfigPurposeOperator)) + } + if name := cnpgOperatorExpand(cnpgOperatorArg(c, "--secret-name"), c, d); name != "" { + add(CNPGOperatorConfigRef{Kind: "Secret", Namespace: d.Namespace, Name: name, Purpose: cnpgConfigPurposeOperator}) + } + if name := cnpgOperatorEnv(c, cnpgMonitoringQueriesCM); name != "" { + add(s.cnpgOperatorConfigMap(r, cache, d.Namespace, name, cnpgConfigPurposeMonitoring)) + } + } + return out +} + +func (s *Server) cnpgOperatorConfigMap(r *http.Request, cache *k8s.ResourceCache, namespace, name, purpose string) CNPGOperatorConfigRef { + ref := CNPGOperatorConfigRef{Kind: "ConfigMap", Namespace: namespace, Name: name, Purpose: purpose} + state := &CNPGOperatorConfigMapState{} + ref.CNPGOperatorConfigMapState = state + if !s.canRead(r, "", "configmaps", namespace, "get") { + state.Reason = "no permission to get ConfigMaps in " + namespace + return ref + } + lister := cache.ConfigMaps() + if lister == nil { + state.Reason = "ConfigMaps are still loading" + return ref + } + if !capacityCacheCoversNamespace(cache, "configmaps", namespace) { + state.Reason = "Radar does not watch ConfigMaps in " + namespace + return ref + } + cm, err := lister.ConfigMaps(namespace).Get(name) + switch { + case apierrors.IsNotFound(err): + exists := false + state.Exists = &exists + state.Reason = "not found" + return ref + case err != nil: + log.Printf("[cnpg] Failed to read ConfigMap %s/%s: %v", namespace, name, err) + state.Reason = "could not read the ConfigMap" + return ref + } + exists := true + state.Exists = &exists + state.Readable = true + state.Data = map[string]string{} + for k, v := range cm.Data { + state.Data[k] = v + } + return ref +} diff --git a/internal/server/cnpg_operator_test.go b/internal/server/cnpg_operator_test.go new file mode 100644 index 000000000..b1f8e49af --- /dev/null +++ b/internal/server/cnpg_operator_test.go @@ -0,0 +1,368 @@ +package server + +import ( + "context" + "encoding/json" + "io" + "net/http" + "testing" + "time" + + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + "github.com/skyhook-io/radar/internal/auth" + "github.com/skyhook-io/radar/internal/k8s" +) + +func cnpgOperatorDeployment() *appsv1.Deployment { + return &appsv1.Deployment{ + ObjectMeta: metav1.ObjectMeta{ + Name: "cnpg-controller-manager", Namespace: "cnpg-system", Generation: 1, + Labels: map[string]string{"app.kubernetes.io/name": "cloudnative-pg"}, + }, + Spec: appsv1.DeploymentSpec{ + Replicas: int32Ptr(1), + Selector: &metav1.LabelSelector{MatchLabels: map[string]string{"app.kubernetes.io/name": "cloudnative-pg"}}, + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{"app.kubernetes.io/name": "cloudnative-pg"}}, + Spec: corev1.PodSpec{Containers: []corev1.Container{ + {Name: "sidecar", Image: "busybox:1.36"}, + { + Name: "manager", + Image: "ghcr.io/cloudnative-pg/cloudnative-pg:1.27.0", + Command: []string{"/manager"}, + Args: []string{ + "controller", "--leader-elect", + "--config-map-name=$(OPERATOR_DEPLOYMENT_NAME)-config", + "--secret-name", "$(OPERATOR_DEPLOYMENT_NAME)-config", + }, + Env: []corev1.EnvVar{ + {Name: "OPERATOR_DEPLOYMENT_NAME", Value: "cnpg-controller-manager"}, + {Name: "MONITORING_QUERIES_CONFIGMAP", Value: "cnpg-default-monitoring"}, + }, + }, + }}, + }, + }, + Status: appsv1.DeploymentStatus{ObservedGeneration: 1, Replicas: 1, ReadyReplicas: 1}, + } +} + +func cnpgPluginDeployment() *appsv1.Deployment { + return &appsv1.Deployment{ + ObjectMeta: metav1.ObjectMeta{Name: "barman-cloud", Namespace: "cnpg-system"}, + Spec: appsv1.DeploymentSpec{ + Selector: &metav1.LabelSelector{MatchLabels: map[string]string{"app": "barman-cloud"}}, + Template: corev1.PodTemplateSpec{ + ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{"app": "barman-cloud"}}, + Spec: corev1.PodSpec{Containers: []corev1.Container{ + {Name: "barman-cloud", Image: "ghcr.io/cloudnative-pg/plugin-barman-cloud:v0.5.0"}, + }}, + }, + }, + } +} + +func cnpgPluginService() *corev1.Service { + return &corev1.Service{ + ObjectMeta: metav1.ObjectMeta{ + Name: "barman-cloud", Namespace: "cnpg-system", + Labels: map[string]string{"cnpg.io/pluginName": "barman-cloud.cloudnative-pg.io"}, + }, + Spec: corev1.ServiceSpec{Selector: map[string]string{"app": "barman-cloud"}}, + } +} + +func cnpgOperatorConfigMapObj() *corev1.ConfigMap { + return &corev1.ConfigMap{ + ObjectMeta: metav1.ObjectMeta{Name: "cnpg-controller-manager-config", Namespace: "cnpg-system"}, + Data: map[string]string{"INHERITED_ANNOTATIONS": "team/*"}, + } +} + +// seedCNPGOperator creates typed objects in the shared fake cluster and waits +// until the cache serves each one. +func seedCNPGOperator(t *testing.T, deployments []*appsv1.Deployment, services []*corev1.Service, configMaps []*corev1.ConfigMap) { + t.Helper() + ctx := context.Background() + for _, d := range deployments { + if _, err := testFakeClient.AppsV1().Deployments(d.Namespace).Create(ctx, d, metav1.CreateOptions{}); err != nil { + t.Fatalf("create deployment %s: %v", d.Name, err) + } + t.Cleanup(func() { + _ = testFakeClient.AppsV1().Deployments(d.Namespace).Delete(context.Background(), d.Name, metav1.DeleteOptions{}) + }) + } + for _, svc := range services { + if _, err := testFakeClient.CoreV1().Services(svc.Namespace).Create(ctx, svc, metav1.CreateOptions{}); err != nil { + t.Fatalf("create service %s: %v", svc.Name, err) + } + t.Cleanup(func() { + _ = testFakeClient.CoreV1().Services(svc.Namespace).Delete(context.Background(), svc.Name, metav1.DeleteOptions{}) + }) + } + for _, cm := range configMaps { + if _, err := testFakeClient.CoreV1().ConfigMaps(cm.Namespace).Create(ctx, cm, metav1.CreateOptions{}); err != nil { + t.Fatalf("create configmap %s: %v", cm.Name, err) + } + t.Cleanup(func() { + _ = testFakeClient.CoreV1().ConfigMaps(cm.Namespace).Delete(context.Background(), cm.Name, metav1.DeleteOptions{}) + }) + } + cache := k8s.GetResourceCache() + deadline := time.Now().Add(5 * time.Second) + for { + missing := 0 + for _, d := range deployments { + if _, err := cache.Deployments().Deployments(d.Namespace).Get(d.Name); err != nil { + missing++ + } + } + for _, svc := range services { + if _, err := cache.Services().Services(svc.Namespace).Get(svc.Name); err != nil { + missing++ + } + } + for _, cm := range configMaps { + if l := cache.ConfigMaps(); l == nil { + missing++ + } else if _, err := l.ConfigMaps(cm.Namespace).Get(cm.Name); err != nil { + missing++ + } + } + if missing == 0 { + return + } + if time.Now().After(deadline) { + t.Fatalf("%d operator fixtures did not reach the cache", missing) + } + time.Sleep(20 * time.Millisecond) + } +} + +func seedFullCNPGOperator(t *testing.T) { + t.Helper() + seedCNPGOperator(t, + []*appsv1.Deployment{cnpgOperatorDeployment(), cnpgPluginDeployment()}, + []*corev1.Service{cnpgPluginService()}, + []*corev1.ConfigMap{cnpgOperatorConfigMapObj()}, + ) +} + +func readCNPGOperator(t *testing.T, resp *http.Response) (CNPGOperatorResponse, []byte) { + t.Helper() + defer resp.Body.Close() + body, err := io.ReadAll(resp.Body) + if err != nil { + t.Fatalf("read body: %v", err) + } + if resp.StatusCode != http.StatusOK { + t.Fatalf("status = %d, want 200: %s", resp.StatusCode, body) + } + var out CNPGOperatorResponse + if err := json.Unmarshal(body, &out); err != nil { + t.Fatalf("decode: %v", err) + } + return out, body +} + +func getCNPGOperatorNoAuth(t *testing.T, query string) (CNPGOperatorResponse, []byte) { + t.Helper() + resp, err := http.Get(testServer.URL + "/api/cnpg/operator" + query) + if err != nil { + t.Fatalf("GET: %v", err) + } + return readCNPGOperator(t, resp) +} + +func findConfigRef(refs []CNPGOperatorConfigRef, kind, purpose string) *CNPGOperatorConfigRef { + for i := range refs { + if refs[i].Kind == kind && refs[i].Purpose == purpose { + return &refs[i] + } + } + return nil +} + +func TestCNPGOperator_DiscoversOperatorPluginAndConfig(t *testing.T) { + seedFullCNPGOperator(t) + + got, body := getCNPGOperatorNoAuth(t, "") + for _, key := range []string{"deployments", "services"} { + if got.Coverage[key].State != cnpgCoverageFull { + t.Errorf("coverage[%s] = %+v, want full", key, got.Coverage[key]) + } + } + if len(got.Components) != 2 { + t.Fatalf("components = %+v, want operator then plugin", got.Components) + } + op, plugin := got.Components[0], got.Components[1] + if op.Role != "operator" || op.Namespace != "cnpg-system" || op.Deployment != "cnpg-controller-manager" || + op.Image != "ghcr.io/cloudnative-pg/cloudnative-pg:1.27.0" || op.Version != "1.27.0" { + t.Errorf("operator = %+v", op) + } + if op.ReadyReplicas == nil || *op.ReadyReplicas != 1 || op.Replicas == nil || *op.Replicas != 1 { + t.Errorf("operator readiness = %v/%v, want 1/1", op.ReadyReplicas, op.Replicas) + } + if plugin.Role != "plugin" || plugin.PluginName != "barman-cloud.cloudnative-pg.io" || plugin.Deployment != "barman-cloud" || plugin.Version != "v0.5.0" { + t.Errorf("plugin = %+v", plugin) + } + if plugin.ReadyReplicas != nil { + t.Errorf("plugin readyReplicas = %d, want null when the controller has reported no status", *plugin.ReadyReplicas) + } + + cm := findConfigRef(got.Config, "ConfigMap", "operator") + if cm == nil || cm.Name != "cnpg-controller-manager-config" || cm.Namespace != "cnpg-system" || cm.CNPGOperatorConfigMapState == nil || + !cm.Readable || cm.Exists == nil || !*cm.Exists || cm.Data["INHERITED_ANNOTATIONS"] != "team/*" { + t.Errorf("operator ConfigMap = %+v", cm) + } + secret := findConfigRef(got.Config, "Secret", "operator") + if secret == nil || secret.Name != "cnpg-controller-manager-config" { + t.Errorf("operator Secret = %+v, want the space-separated --secret-name resolved", secret) + } + mon := findConfigRef(got.Config, "ConfigMap", "monitoring") + if mon == nil || mon.Name != "cnpg-default-monitoring" || mon.Readable || mon.Exists == nil || *mon.Exists { + t.Errorf("monitoring ConfigMap = %+v, want exists=false readable=false", mon) + } + + var raw struct { + Config []map[string]any `json:"config"` + Components []map[string]any `json:"components"` + } + if err := json.Unmarshal(body, &raw); err != nil { + t.Fatalf("raw decode: %v", err) + } + for _, ref := range raw.Config { + if ref["kind"] != "Secret" { + continue + } + for _, k := range []string{"data", "exists", "readable", "keys"} { + if _, ok := ref[k]; ok { + t.Errorf("Secret reference carries %q: %v", k, ref) + } + } + } + if v, ok := raw.Components[1]["readyReplicas"]; !ok || v != nil { + t.Errorf("plugin readyReplicas JSON = %v (present=%v), want explicit null", v, ok) + } +} + +func TestCNPGOperator_VersionFallsBackToLabelAndNeverInvents(t *testing.T) { + digest := cnpgOperatorDeployment() + digest.Name = "pinned" + digest.Labels["app.kubernetes.io/version"] = "1.26.1" + digest.Spec.Template.Spec.Containers[1].Image = "ghcr.io/cloudnative-pg/cloudnative-pg@sha256:abc" + bare := cnpgOperatorDeployment() + bare.Name = "bare" + bare.Spec.Template.Spec.Containers[1].Image = "ghcr.io/cloudnative-pg/cloudnative-pg" + seedCNPGOperator(t, []*appsv1.Deployment{digest, bare}, nil, nil) + + got, _ := getCNPGOperatorNoAuth(t, "") + versions := map[string]string{} + for _, c := range got.Components { + versions[c.Deployment] = c.Version + } + if versions["pinned"] != "1.26.1" { + t.Errorf("digest-pinned version = %q, want the version label", versions["pinned"]) + } + if v, ok := versions["bare"]; !ok || v != "" { + t.Errorf("untagged, unlabelled version = %q (found=%v), want empty", v, ok) + } +} + +func TestCNPGOperator_IgnoresNamespaceViewFilter(t *testing.T) { + seedFullCNPGOperator(t) + got, _ := getCNPGOperatorNoAuth(t, "?namespaces=default") + if len(got.Components) == 0 || got.Components[0].Deployment != "cnpg-controller-manager" { + t.Errorf("components = %+v, want the operator in cnpg-system despite a view filter on default", got.Components) + } + + env := newAuthTestServer(t) + perms := &auth.UserPermissions{AllowedNamespaces: []string{"cnpg-system", "default"}} + allow(perms, "apps", "deployments", "", true) + allow(perms, "", "services", "", true) + env.srv.permCache.Set("viewer", nil, perms) + authed, _ := readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator?namespaces=default", "viewer", "")) + if len(authed.Components) == 0 || authed.Components[0].Namespace != "cnpg-system" { + t.Errorf("auth components = %+v, want the operator despite the view filter", authed.Components) + } +} + +func TestCNPGOperator_ConfigMapDataNeedsGet(t *testing.T) { + seedFullCNPGOperator(t) + env := newAuthTestServer(t) + for _, u := range []struct { + name string + getCM bool + }{{"reads-cm", true}, {"no-cm", false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"cnpg-system"}} + allow(perms, "apps", "deployments", "", true) + allow(perms, "", "services", "", true) + perms.SetCanI("get", "", "configmaps", "cnpg-system", u.getCM) + env.srv.permCache.Set(u.name, nil, perms) + } + + got, _ := readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator", "reads-cm", "")) + cm := findConfigRef(got.Config, "ConfigMap", "operator") + if cm == nil || cm.CNPGOperatorConfigMapState == nil || !cm.Readable || cm.Data["INHERITED_ANNOTATIONS"] != "team/*" { + t.Errorf("with get configmaps: %+v", cm) + } + + got, body := readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator", "no-cm", "")) + cm = findConfigRef(got.Config, "ConfigMap", "operator") + if cm == nil || cm.CNPGOperatorConfigMapState == nil || cm.Readable || cm.Exists != nil || cm.Data != nil || cm.Reason == "" { + t.Errorf("without get configmaps: %+v, want unreadable, existence unknown, no data, a reason", cm) + } + var raw struct { + Config []map[string]any `json:"config"` + } + _ = json.Unmarshal(body, &raw) + for _, ref := range raw.Config { + if ref["kind"] == "ConfigMap" && ref["purpose"] == "operator" && ref["data"] != nil { + t.Errorf("ConfigMap data returned without get: %v", ref) + } + } +} + +func TestCNPGOperator_DeniedDeploymentsWithholdComponents(t *testing.T) { + seedFullCNPGOperator(t) + env := newAuthTestServer(t) + + partial := &auth.UserPermissions{AllowedNamespaces: []string{"cnpg-system", "default"}} + allow(partial, "apps", "deployments", "", false) + allow(partial, "apps", "deployments", "cnpg-system", false) + allow(partial, "apps", "deployments", "default", true) + allow(partial, "", "services", "", true) + env.srv.permCache.Set("partial", nil, partial) + + got, _ := readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator", "partial", "")) + cov := got.Coverage["deployments"] + if cov.State != cnpgCoveragePartial || len(cov.DeniedNamespaces) != 1 || cov.DeniedNamespaces[0] != "cnpg-system" { + t.Errorf("deployments coverage = %+v, want partial denied [cnpg-system]", cov) + } + for _, c := range got.Components { + if c.Namespace == "cnpg-system" { + t.Errorf("component from a namespace whose Deployments are denied: %+v", c) + } + } + if len(got.Config) != 0 { + t.Errorf("config = %+v, want none without a visible operator", got.Config) + } + + none := &auth.UserPermissions{AllowedNamespaces: []string{"cnpg-system"}} + allow(none, "apps", "deployments", "", false) + allow(none, "apps", "deployments", "cnpg-system", false) + allow(none, "", "services", "", false) + allow(none, "", "services", "cnpg-system", false) + env.srv.permCache.Set("none", nil, none) + + got, _ = readCNPGOperator(t, env.authGet(t, "/api/cnpg/operator", "none", "")) + if got.Coverage["deployments"].State != cnpgCoverageDenied || got.Coverage["services"].State != cnpgCoverageDenied { + t.Errorf("coverage = %+v, want both denied", got.Coverage) + } + if got.Components == nil || len(got.Components) != 0 || got.Config == nil { + t.Errorf("components=%v config=%v, want empty arrays", got.Components, got.Config) + } +} diff --git a/internal/server/cnpg_workspace.go b/internal/server/cnpg_workspace.go index 58ff2e7db..1421e0218 100644 --- a/internal/server/cnpg_workspace.go +++ b/internal/server/cnpg_workspace.go @@ -68,10 +68,25 @@ var cnpgWorkspaceKinds = []cnpgWorkspaceKind{ } // CNPGWorkspaceCoverage states how much of one kind the caller could see. -// DeniedNamespaces lists only namespaces already in the caller's scope. +// DeniedNamespaces lists only namespaces already in the caller's scope, so it +// may be omitted on a partial state; AllowedNamespaces is always set on a +// partial state and is the authority for which namespaces were read. type CNPGWorkspaceCoverage struct { - State string `json:"state"` - DeniedNamespaces []string `json:"deniedNamespaces,omitempty"` + State string `json:"state"` + DeniedNamespaces []string `json:"deniedNamespaces,omitempty"` + AllowedNamespaces []string `json:"allowedNamespaces,omitempty"` +} + +func cnpgCoverageOf(acc cnpgKindAccess, denied []string) CNPGWorkspaceCoverage { + cov := CNPGWorkspaceCoverage{State: acc.state, DeniedNamespaces: denied} + if acc.state == cnpgCoveragePartial { + cov.AllowedNamespaces = make([]string, 0, len(acc.namespaces)) + for ns := range acc.namespaces { + cov.AllowedNamespaces = append(cov.AllowedNamespaces, ns) + } + sort.Strings(cov.AllowedNamespaces) + } + return cov } // CNPGWorkspaceIssue is the subset of issuesapi.Issue the workspace renders. @@ -192,7 +207,7 @@ func (s *Server) handleCNPGWorkspace(w http.ResponseWriter, r *http.Request) { } access[k.key] = acc items[k.key] = list - resp.Coverage[k.key] = CNPGWorkspaceCoverage{State: acc.state, DeniedNamespaces: denied} + resp.Coverage[k.key] = cnpgCoverageOf(acc, denied) } if !resp.Installed { s.writeJSON(w, resp) @@ -215,48 +230,55 @@ func (s *Server) handleCNPGWorkspace(w http.ResponseWriter, r *http.Request) { resp.Objects[k.key] = out } - podAccess, podDenied, pods := s.cnpgWorkspaceReadPods(r, cache, namespaces) + podAccess, podDenied, pods, instancePods := s.cnpgWorkspaceReadPods(r, cache, namespaces, cnpgClusterUIDs(items[cnpgWorkspaceClusterKey])) access[cnpgWorkspacePodsKey] = podAccess - resp.Coverage[cnpgWorkspacePodsKey] = CNPGWorkspaceCoverage{State: podAccess.state, DeniedNamespaces: podDenied} + resp.Coverage[cnpgWorkspacePodsKey] = cnpgCoverageOf(podAccess, podDenied) resp.Objects[cnpgWorkspacePodsKey] = pods - resp.Issues = s.cnpgWorkspaceIssues(r, namespaces, access) + resp.Issues = s.cnpgWorkspaceIssues(r, namespaces, access, instancePods) resp.Audit = cnpgWorkspaceAudit(items[cnpgWorkspaceClusterKey], items[cnpgWorkspaceSchedKey], access[cnpgWorkspaceSchedKey]) s.writeJSON(w, resp) } // cnpgWorkspaceScope resolves where the caller may list one namespaced -// resource: nil allowed means the whole request scope. denied only ever names -// namespaces drawn from the caller's own scope (their view filter, or all -// namespaces for a caller who may see every namespace). -func (s *Server) cnpgWorkspaceScope(r *http.Request, namespaces []string, group, resource string) (allowed, denied []string, any bool) { +// resource: nil allowed means the whole request scope. +// +// denied names namespaces only when the candidate set came from the caller — +// their view filter or their RBAC-allowed list. When the scope is "all" the +// candidates are every namespace in Radar's cache, and naming the denied ones +// would disclose namespaces the caller was never shown; partial then carries +// the fact without the names. +func (s *Server) cnpgWorkspaceScope(r *http.Request, namespaces []string, group, resource string) (allowed, denied []string, partial, any bool) { if noNamespaceAccess(namespaces) { - return []string{}, nil, false + return []string{}, nil, false, false } if s.canRead(r, group, resource, "", "list") { - return namespaces, nil, true + return namespaces, nil, false, true } candidates := namespaces if candidates == nil { candidates = allNamespaceNames() } if len(candidates) == 0 { - return []string{}, nil, false + return []string{}, nil, false, false } allowed = s.filterNamespacesByCanRead(r, group, resource, "list", candidates) - for _, ns := range candidates { - if !slices.Contains(allowed, ns) { - denied = append(denied, ns) + partial = len(allowed) < len(candidates) + if namespaces != nil { + for _, ns := range candidates { + if !slices.Contains(allowed, ns) { + denied = append(denied, ns) + } } + sort.Strings(denied) } - sort.Strings(denied) - return allowed, denied, len(allowed) > 0 + return allowed, denied, partial, len(allowed) > 0 } -func accessFromScope(allowed, denied []string) cnpgKindAccess { +func accessFromScope(allowed []string, partial bool) cnpgKindAccess { acc := cnpgKindAccess{state: cnpgCoverageFull, all: allowed == nil} - if len(denied) > 0 { + if partial { acc.state = cnpgCoveragePartial } if allowed != nil { @@ -277,11 +299,11 @@ func (s *Server) cnpgWorkspaceReadKind(r *http.Request, cache *k8s.ResourceCache } acc = cnpgKindAccess{state: cnpgCoverageFull, all: true} } else { - allowed, d, ok := s.cnpgWorkspaceScope(r, namespaces, k.group, k.resource) + allowed, d, partial, ok := s.cnpgWorkspaceScope(r, namespaces, k.group, k.resource) if !ok { return cnpgKindAccess{state: cnpgCoverageDenied}, nil, nil } - acc, denied, readNamespaces = accessFromScope(allowed, d), d, allowed + acc, denied, readNamespaces = accessFromScope(allowed, partial), d, allowed } list, err := readCNPGKind(r.Context(), cache, k, readNamespaces) @@ -429,15 +451,21 @@ type cnpgWorkspacePod struct { } `json:"status"` } -// isCNPGInstancePod requires the controller-set ownerReference as well as the -// label: a label alone is something any workload can carry. -func isCNPGInstancePod(p *corev1.Pod) bool { +// isCNPGInstancePod requires the controller ownerReference to name a visible +// Cluster by UID, not just by name: a label alone is something any workload +// can carry, and a Pod left behind by a deleted Cluster must not be attributed +// to a new one created under the same name. clusterUIDs is keyed ns/name. +func isCNPGInstancePod(p *corev1.Pod, clusterUIDs map[string]types.UID) bool { clusterName := p.Labels["cnpg.io/cluster"] if clusterName == "" { return false } + uid, ok := clusterUIDs[p.Namespace+"/"+clusterName] + if !ok || uid == "" { + return false + } for _, ref := range p.OwnerReferences { - if ref.Kind != "Cluster" || ref.Name != clusterName { + if ref.Controller == nil || !*ref.Controller || ref.Kind != "Cluster" || ref.Name != clusterName || ref.UID != uid { continue } if gv, err := schema.ParseGroupVersion(ref.APIVersion); err == nil && gv.Group == cnpgGroup { @@ -447,6 +475,14 @@ func isCNPGInstancePod(p *corev1.Pod) bool { return false } +func cnpgClusterUIDs(clusters []*unstructured.Unstructured) map[string]types.UID { + out := make(map[string]types.UID, len(clusters)) + for _, c := range clusters { + out[c.GetNamespace()+"/"+c.GetName()] = c.GetUID() + } + return out +} + func trimCNPGPod(p *corev1.Pod) cnpgWorkspacePod { out := cnpgWorkspacePod{APIVersion: "v1", Kind: "Pod"} out.Metadata = cnpgWorkspacePodMeta{ @@ -470,37 +506,53 @@ func trimCNPGPod(p *corev1.Pod) cnpgWorkspacePod { return out } -func (s *Server) cnpgWorkspaceReadPods(r *http.Request, cache *k8s.ResourceCache, namespaces []string) (cnpgKindAccess, []string, []any) { - out := []any{} - allowed, denied, ok := s.cnpgWorkspaceScope(r, namespaces, "", "pods") +// cnpgTypedScope resolves where the caller may list a typed kind and which of +// those namespaces Radar's informer actually holds. The informer may itself be +// namespace-scoped when Radar's own identity cannot list the kind +// cluster-wide; what it does not hold is unread, not empty. read is nil for +// "every namespace". +func (s *Server) cnpgTypedScope(r *http.Request, cache *k8s.ResourceCache, namespaces []string, group, resource string) (acc cnpgKindAccess, denied, read []string) { + allowed, denied, partial, ok := s.cnpgWorkspaceScope(r, namespaces, group, resource) if !ok { - return cnpgKindAccess{state: cnpgCoverageDenied}, nil, out - } - if cache.Pods() == nil { - log.Printf("[cnpg] Pod cache unavailable for workspace") - return cnpgKindAccess{state: cnpgCoverageError}, nil, out + return cnpgKindAccess{state: cnpgCoverageDenied}, nil, []string{} } - // The typed Pod informer may itself be namespace-scoped when Radar's own - // identity cannot list Pods cluster-wide; what it does not hold is unread. - within := capacityNamespacesWithinCache(cache, "pods", allowed) + within := capacityNamespacesWithinCache(cache, resource, allowed) if within.unavailable { - log.Printf("[cnpg] Pod cache does not cover the workspace scope") - return cnpgKindAccess{state: cnpgCoverageError}, nil, out + log.Printf("[cnpg] %s cache does not cover the requested scope", resource) + return cnpgKindAccess{state: cnpgCoverageError}, nil, []string{} } if allowed != nil { for _, ns := range allowed { - if !slices.Contains(within.namespaces, ns) { + if slices.Contains(within.namespaces, ns) { + continue + } + partial = true + if namespaces != nil { denied = append(denied, ns) } } sort.Strings(denied) } - acc := accessFromScope(within.namespaces, denied) - if within.partial { - acc.state = cnpgCoveragePartial + acc = accessFromScope(within.namespaces, partial || within.partial) + return acc, denied, within.namespaces +} + +// cnpgWorkspaceReadPods returns the instance Pods of visible Clusters, plus +// the namespace/name set of what it returned — the only Pods whose issues the +// response may carry. +func (s *Server) cnpgWorkspaceReadPods(r *http.Request, cache *k8s.ResourceCache, namespaces []string, clusterUIDs map[string]types.UID) (cnpgKindAccess, []string, []any, map[string]bool) { + out := []any{} + returned := map[string]bool{} + acc, denied, read := s.cnpgTypedScope(r, cache, namespaces, "", "pods") + if acc.state == cnpgCoverageDenied || acc.state == cnpgCoverageError { + return acc, nil, out, returned + } + if cache.Pods() == nil { + log.Printf("[cnpg] Pod cache unavailable for workspace") + return cnpgKindAccess{state: cnpgCoverageError}, nil, out, returned } - pods := listPodsScoped(cache.Pods(), within.namespaces) + pods := listPodsScoped(cache.Pods(), read) sort.Slice(pods, func(i, j int) bool { if pods[i].Namespace != pods[j].Namespace { return pods[i].Namespace < pods[j].Namespace @@ -508,11 +560,12 @@ func (s *Server) cnpgWorkspaceReadPods(r *http.Request, cache *k8s.ResourceCache return pods[i].Name < pods[j].Name }) for _, p := range pods { - if p != nil && isCNPGInstancePod(p) { + if p != nil && isCNPGInstancePod(p, clusterUIDs) { out = append(out, trimCNPGPod(p)) + returned[p.Namespace+"/"+p.Name] = true } } - return acc, denied, out + return acc, denied, out, returned } var cnpgWorkspaceKeyByGroupKind = func() map[string]string { @@ -523,10 +576,13 @@ var cnpgWorkspaceKeyByGroupKind = func() map[string]string { return m }() -// cnpgWorkspaceIssues runs the same composition /api/issues serves, then keeps -// only CNPG subjects on a kind and namespace this response had coverage for — -// an issue on an object the caller could not list would disclose it. -func (s *Server) cnpgWorkspaceIssues(r *http.Request, namespaces []string, access map[string]cnpgKindAccess) []CNPGWorkspaceIssue { +// cnpgWorkspaceIssues runs the same composition /api/issues serves, but reads +// the flat evidence rows: the grouped view folds instance-Pod evidence into +// the owning Cluster's row, which would hand Pod failure detail to a caller +// who may list Clusters but not Pods. A row is kept only when its own subject +// is visible here — a CNPG kind covered in its namespace, or an instance Pod +// this response returned. IDs are the subject-derived IDs /api/issues uses. +func (s *Server) cnpgWorkspaceIssues(r *http.Request, namespaces []string, access map[string]cnpgKindAccess, instancePods map[string]bool) []CNPGWorkspaceIssue { out := []CNPGWorkspaceIssue{} if noNamespaceAccess(namespaces) { return out @@ -538,16 +594,11 @@ func (s *Server) cnpgWorkspaceIssues(r *http.Request, namespaces []string, acces composed, _ := issues.ComposeWithStats(provider, issues.Filters{ Namespaces: namespaces, Limit: issues.NoLimit, - Grouped: true, CanReadClusterScoped: s.issueClusterScopedAccess(r), CanReadRelated: s.issueRelatedResourceAccess(r), }) for _, iss := range composed { - if iss.Group != cnpgGroup && iss.Group != cnpgBarmanGroup { - continue - } - key, ok := cnpgWorkspaceKeyByGroupKind[iss.Group+"/"+iss.Kind] - if !ok || !access[key].covers(iss.Namespace) { + if !cnpgWorkspaceIssueVisible(iss, access, instancePods) { continue } out = append(out, CNPGWorkspaceIssue{ @@ -568,6 +619,17 @@ func (s *Server) cnpgWorkspaceIssues(r *http.Request, namespaces []string, acces return out } +func cnpgWorkspaceIssueVisible(iss issues.Issue, access map[string]cnpgKindAccess, instancePods map[string]bool) bool { + if iss.Group == "" && iss.Kind == "Pod" { + return access[cnpgWorkspacePodsKey].covers(iss.Namespace) && instancePods[iss.Namespace+"/"+iss.Name] + } + if iss.Group != cnpgGroup && iss.Group != cnpgBarmanGroup { + return false + } + key, ok := cnpgWorkspaceKeyByGroupKind[iss.Group+"/"+iss.Kind] + return ok && access[key].covers(iss.Namespace) +} + // cnpgWorkspaceAudit reports the declarative-backup posture finding only for // Clusters whose namespace had its ScheduledBackups read: without that list, // "no schedule targets this cluster" is an absence nobody established. diff --git a/internal/server/cnpg_workspace_test.go b/internal/server/cnpg_workspace_test.go index ee2d4edf3..02152adb1 100644 --- a/internal/server/cnpg_workspace_test.go +++ b/internal/server/cnpg_workspace_test.go @@ -4,6 +4,7 @@ import ( "context" "encoding/json" "net/http" + "strings" "testing" "time" @@ -12,6 +13,7 @@ import ( "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/apimachinery/pkg/types" dynamicfake "k8s.io/client-go/dynamic/fake" "github.com/skyhook-io/radar/internal/auth" @@ -193,7 +195,7 @@ func cnpgPod(ns, name, clusterLabel string, owners ...metav1.OwnerReference) *co func TestCNPGWorkspace_AuthDisabledReturnsEverythingAndOnlyOwnedInstancePods(t *testing.T) { seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, - cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgws", "pg-orders", map[string]any{"instances": int64(1)}, nil), + withUID(cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgws", "pg-orders", map[string]any{"instances": int64(1)}, nil), "c-uid"), cnpgObj("postgresql.cnpg.io/v1", "Pooler", "pgws", "pg-orders-rw", map[string]any{"cluster": map[string]any{"name": "pg-orders"}}, nil), cnpgObj("postgresql.cnpg.io/v1", "ClusterImageCatalog", "", "pg-fleet", nil, nil), cnpgObj("barmancloud.cnpg.io/v1", "ObjectStore", "pgws", "store", nil, nil), @@ -204,6 +206,8 @@ func TestCNPGWorkspace_AuthDisabledReturnsEverythingAndOnlyOwnedInstancePods(t * cnpgPod("pgws", "impostor-1", "pg-orders"), cnpgPod("pgws", "capi-owned-1", "pg-orders", metav1.OwnerReference{APIVersion: "cluster.x-k8s.io/v1beta1", Kind: "Cluster", Name: "pg-orders", UID: "x"}), cnpgPod("pgws", "other-owner-1", "pg-orders", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-billing", UID: "y"}), + cnpgPod("pgws", "stale-uid-1", "pg-orders", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "deleted-cluster-uid", Controller: boolPtr(true)}), + cnpgPod("pgws", "not-controller-1", "pg-orders", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "c-uid"}), ) got := getWorkspaceNoAuth(t, "") @@ -333,6 +337,9 @@ func TestCNPGWorkspace_PartialNamespaceCoverage(t *testing.T) { if cov.State != cnpgCoveragePartial || len(cov.DeniedNamespaces) != 1 || cov.DeniedNamespaces[0] != "b" { t.Errorf("clusters coverage = %+v, want partial denied [b]", cov) } + if len(cov.AllowedNamespaces) != 1 || cov.AllowedNamespaces[0] != "a" { + t.Errorf("allowedNamespaces = %v, want [a]", cov.AllowedNamespaces) + } names := objectNames(got.Objects["clusters"]) if len(names) != 1 || names[0] != "pg-a" { t.Errorf("clusters = %v, want only pg-a", names) @@ -451,14 +458,127 @@ func TestCNPGWorkspace_AuditNeedsScheduledBackupEvidence(t *testing.T) { } } +func withUID(u *unstructured.Unstructured, uid string) *unstructured.Unstructured { + u.SetUID(types.UID(uid)) + return u +} + func TestIsCNPGInstancePod(t *testing.T) { - owned := cnpgPod("pg", "x-1", "x", metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "x"}) - if !isCNPGInstancePod(owned) { - t.Error("owned instance pod rejected") - } - unlabeled := owned.DeepCopy() - unlabeled.Labels = nil - if isCNPGInstancePod(unlabeled) { - t.Error("pod without the cluster label accepted") + uids := map[string]types.UID{"pg/x": "x-uid"} + ref := func(uid string, controller bool) metav1.OwnerReference { + return metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "x", UID: types.UID(uid), Controller: boolPtr(controller)} + } + for _, c := range []struct { + name string + pod *corev1.Pod + uids map[string]types.UID + want bool + }{ + {"controller ref to the visible Cluster", cnpgPod("pg", "x-1", "x", ref("x-uid", true)), uids, true}, + {"left behind by a deleted Cluster of the same name", cnpgPod("pg", "x-1", "x", ref("old-uid", true)), uids, false}, + {"non-controller owner", cnpgPod("pg", "x-1", "x", ref("x-uid", false)), uids, false}, + {"Cluster not visible", cnpgPod("pg", "x-1", "x", ref("x-uid", true)), map[string]types.UID{}, false}, + {"Cluster of that name in another namespace", cnpgPod("other", "x-1", "x", ref("x-uid", true)), uids, false}, + {"no cluster label", func() *corev1.Pod { p := cnpgPod("pg", "x-1", "x", ref("x-uid", true)); p.Labels = nil; return p }(), uids, false}, + } { + t.Run(c.name, func(t *testing.T) { + if got := isCNPGInstancePod(c.pod, c.uids); got != c.want { + t.Errorf("isCNPGInstancePod = %v, want %v", got, c.want) + } + }) + } +} + +// The grouped issue view folds instance-Pod evidence into the owning Cluster's +// row. A caller who may list Clusters but not Pods must not receive that +// evidence in any form; one who may list Pods receives it on the Pod itself. +func TestCNPGWorkspace_PodEvidenceFollowsPodAccess(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + withUID(cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgev", "pg-orders", map[string]any{"instances": int64(1)}, nil), "orders-uid"), + ) + crashing := cnpgPod("pgev", "pg-orders-1", "pg-orders", metav1.OwnerReference{ + APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "orders-uid", Controller: boolPtr(true), + }) + crashing.Status.ContainerStatuses[0] = corev1.ContainerStatus{ + Name: "postgres", Ready: false, RestartCount: 9, Image: "pg:17", + State: corev1.ContainerState{Waiting: &corev1.ContainerStateWaiting{Reason: "CrashLoopBackOff", Message: "back-off restarting failed container"}}, + } + seedCNPGPods(t, crashing) + + env := newAuthTestServer(t) + for _, u := range []struct { + name string + pods bool + }{{"with-pods", true}, {"clusters-only", false}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pgev"}} + allow(perms, cnpgGroup, "clusters", "", true) + allow(perms, "", "pods", "", u.pods) + allow(perms, "", "pods", "pgev", u.pods) + env.srv.permCache.Set(u.name, nil, perms) + } + + control := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "with-pods", "")) + var podIssue *CNPGWorkspaceIssue + for i, iss := range control.Issues { + if iss.Kind == "Pod" && iss.Name == "pg-orders-1" { + podIssue = &control.Issues[i] + } + } + if podIssue == nil { + t.Fatalf("with pod access: no Pod issue for the crashlooping instance, got %+v", control.Issues) + } + if podIssue.ID == "" { + t.Error("Pod issue carries no ID") + } + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "clusters-only", "")) + if got.Coverage["pods"].State != cnpgCoverageDenied || len(got.Objects["pods"]) != 0 { + t.Errorf("pods coverage=%+v objects=%v, want denied and []", got.Coverage["pods"], objectNames(got.Objects["pods"])) + } + for _, iss := range got.Issues { + if iss.Kind == "Pod" || strings.Contains(iss.Message, "CrashLoopBackOff") || strings.Contains(iss.Message, "back-off") || iss.ID == podIssue.ID { + t.Errorf("Pod evidence reached a caller without Pod access: %+v", iss) + } + } +} + +// Denied namespaces are named only when the caller supplied the candidate set. +// For a caller whose scope is "all", the candidates are every namespace Radar +// holds, and listing the denied ones would disclose them. +func TestCNPGWorkspace_DeniedNamespacesNeverComeFromTheServerInventory(t *testing.T) { + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + cnpgObj("postgresql.cnpg.io/v1", "Cluster", "default", "pg-default", nil, nil), + ) + env := newAuthTestServer(t) + perms := &auth.UserPermissions{} + allow(perms, cnpgGroup, "clusters", "", false) + for _, ns := range allNamespaceNames() { + allow(perms, cnpgGroup, "clusters", ns, ns == "default") + } + allow(perms, cnpgGroup, "clusters", "broken", false) + env.srv.permCache.Set("wide", nil, perms) + + got := decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace", "wide", "")) + cov := got.Coverage["clusters"] + if cov.State != cnpgCoveragePartial { + t.Errorf("unfiltered: clusters coverage = %+v, want partial", cov) + } + if len(cov.DeniedNamespaces) != 0 { + t.Errorf("unfiltered: deniedNamespaces = %v, want omitted — they came from Radar's namespace inventory", cov.DeniedNamespaces) + } + if len(cov.AllowedNamespaces) != 1 || cov.AllowedNamespaces[0] != "default" { + t.Errorf("unfiltered: allowedNamespaces = %v, want [default] — without it the reader cannot tell which namespaces were read", cov.AllowedNamespaces) + } + if !containsName(got.Objects["clusters"], "pg-default") { + t.Errorf("unfiltered: clusters = %v, want pg-default", objectNames(got.Objects["clusters"])) + } + + got = decodeWorkspace(t, env.authGet(t, "/api/cnpg/workspace?namespaces=default,broken", "wide", "")) + cov = got.Coverage["clusters"] + if cov.State != cnpgCoveragePartial || len(cov.DeniedNamespaces) != 1 || cov.DeniedNamespaces[0] != "broken" { + t.Errorf("filtered: clusters coverage = %+v, want partial naming broken", cov) + } + if len(cov.AllowedNamespaces) != 1 || cov.AllowedNamespaces[0] != "default" { + t.Errorf("filtered: allowedNamespaces = %v, want [default]", cov.AllowedNamespaces) } } diff --git a/internal/server/server.go b/internal/server/server.go index 3fde11d61..1cf9f20b0 100644 --- a/internal/server/server.go +++ b/internal/server/server.go @@ -592,6 +592,7 @@ func (s *Server) setupAppRoutes(r chi.Router) { r.Get("/rbac/namespace/{namespace}", s.handleRBACNamespace) r.Get("/rbac/whoami", s.handleRBACWhoami) r.Get("/cnpg/workspace", s.handleCNPGWorkspace) + r.Get("/cnpg/operator", s.handleCNPGOperator) r.Get("/cnpg/imagecatalogs/{namespace}/{name}/clusters", s.handleCNPGCatalogUsers) r.Get("/cnpg/clusterimagecatalogs/{name}/clusters", s.handleCNPGCatalogUsers) r.Get("/velero/backupstoragelocations/{namespace}/{name}/backups", s.handleVeleroStoredBackups) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx new file mode 100644 index 000000000..beb3580cb --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx @@ -0,0 +1,211 @@ +import { cronToHuman, formatDuration } from '../resources/resource-utils' +import { + CNPG_BARMAN_OBJECTSTORE_GROUP, + CNPG_GROUP, + getCNPGBackupStatus, + getCNPGScheduledBackupNextSchedule, + getCNPGScheduledBackupStatus, +} from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' +import { ClusterLink, NotReported, Note, ObjectProblems, PhaseBadge, SummaryShell, TimeAgo } from './CNPGSharedSummary' +import { + backupDestination, + backupsForScheduledBackup, + clustersIn, + refOf, + relationUnavailable, + scheduledBackupOf, + workspaceList, +} from './relations' + +interface SummaryProps { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +} + +const RECENT_RUNS = 5 + +function methodText(resource: any): string | null { + const m = resource?.status?.method || resource?.spec?.method + if (!m) return null + switch (m) { + case 'plugin': + return resource?.spec?.pluginConfiguration?.name ? `Plugin · ${resource.spec.pluginConfiguration.name}` : 'Plugin' + case 'barmanObjectStore': + return 'Barman object store (in-tree)' + case 'volumeSnapshot': + return 'Volume snapshot' + default: + return m + } +} + +function Timing({ resource }: { resource: any }) { + const started = resource?.status?.startedAt + const stopped = resource?.status?.stoppedAt + const startMs = started ? Date.parse(started) : NaN + const stopMs = stopped ? Date.parse(stopped) : NaN + const tookMs = Number.isFinite(startMs) && Number.isFinite(stopMs) && stopMs >= startMs ? stopMs - startMs : null + return ( + <> + + + + + {stopped ? ( + + + {tookMs !== null && · took {formatDuration(tookMs, true)}} + + ) : Number.isFinite(startMs) ? ( + Not stopped · running for {formatDuration(Date.now() - startMs, true)} + ) : ( + + )} + + + ) +} + +export function CNPGBackupSummary({ resource, workspace, onNavigate }: SummaryProps) { + const ns = resource?.metadata?.namespace ?? '' + const status = getCNPGBackupStatus(resource) + const pod = resource?.status?.instanceID?.podName + const error = resource?.status?.error + const trigger = scheduledBackupOf(resource) + const dest = backupDestination(resource, clustersIn(workspace)) + const method = methodText(resource) + + return ( + + + + Outcome + + + + + + + {pod ? : } + + + {resource?.status?.backupId ? {resource.status.backupId} : } + + + {resource?.spec?.target ? resource.spec.target : Not set · the Cluster's backup target applies} + + {error && ( + + {error} + + )} + + + Relationships + + + + + + {trigger ? ( + + ScheduledBackup{' '} + + + ) : ( + 'On demand' + )} + + + {dest.type === 'objectStore' ? ( + + ObjectStore{' '} + + + ) : dest.type === 'path' ? ( + {dest.path} + ) : dest.type === 'volumeSnapshot' ? ( + 'Volume snapshot' + ) : ( + + )} + + {method ?? } + + + ) +} + +export function CNPGScheduledBackupSummary({ resource, workspace, onNavigate }: SummaryProps) { + const ns = resource?.metadata?.namespace ?? '' + const cron = resource?.spec?.schedule + const human = cron ? cronToHuman(cron) : '' + const next = getCNPGScheduledBackupNextSchedule(resource) + const runsUnavailable = relationUnavailable(workspace, 'backups', ns, 'Backups') + const runs = runsUnavailable ? [] : backupsForScheduledBackup(resource, workspaceList(workspace, 'backups')) + const shown = runs.slice(0, RECENT_RUNS) + + return ( + + + + Schedule + + + + + + + + + {cron ? ( + + {cron} + {human && human !== cron && · {human}} + + ) : ( + + )} + + + + + {next === '-' ? : next} + {methodText(resource) ?? 'Barman object store (in-tree) · default'} + + + shown.length ? `${shown.length} of ${runs.length}` : undefined}>Recent runs + {runsUnavailable ? ( +
+ +
+ ) : shown.length === 0 ? ( +
No Backups from this schedule are visible
+ ) : ( + + {shown.map((b) => ( + } + > + + + {b.status?.startedAt ? ( + + started + + ) : ( + + )} + + + ))} + + )} + {!runsUnavailable && (workspace?.backupsOmitted ?? 0) > 0 && Backups older than 7 days are not listed.} +
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx index 1bf30a1f5..712937a2f 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx @@ -144,7 +144,9 @@ export function CNPGClusterSummary({ {row.poolers.length === 0 ? ( - None + + {row.poolersKnown ? 'None' : 'No access to Poolers'} + ) : ( {row.poolers.map((name) => ( diff --git a/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx new file mode 100644 index 000000000..395c05ca4 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx @@ -0,0 +1,264 @@ +import type { ReactNode } from 'react' +import { getCNPGDeclarativeMessage, getCNPGReclaimPolicy } from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' +import { ClusterLink, NotReported, ObjectProblems, SummaryShell } from './CNPGSharedSummary' +import { + appliedFact, + clustersIn, + databaseForDeclaration, + gitopsSourceOf, + missingManagedRole, + observedGenerationFact, + refOf, + relationUnavailable, + replicationForDatabase, + targetCluster, + workspaceList, +} from './relations' + +interface SummaryProps { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +} + +function ReclaimRow({ resource }: { resource: any }) { + const reclaim = getCNPGReclaimPolicy(resource) + return ( + + {reclaim.destructive ? ( + {reclaim.value} · removing this resource drops it from PostgreSQL + ) : ( + {reclaim.value} · removing this resource leaves PostgreSQL untouched + )} + + ) +} + +function Reconciled({ resource, extra }: { resource: any; extra?: ReactNode }) { + const message = getCNPGDeclarativeMessage(resource) + const applied = appliedFact(resource) + return ( + <> + Reconciled + + + + + + + + + {message ? ( + {message} + ) : ( + + )} + + {extra} + + + ) +} + +function DeclaredIn({ resource }: { resource: any }) { + const src = gitopsSourceOf(resource) + if (!src) return Applied directly (no GitOps owner label) + return ( + + {src.tool === 'argocd' ? 'Argo CD application' : 'Flux'} {src.namespace ? `${src.namespace}/${src.name}` : src.name} + + ) +} + +function DatabaseRef({ resource, workspace, onNavigate }: SummaryProps) { + const dbname = resource?.spec?.dbname + if (!dbname) return + const ns = resource?.metadata?.namespace ?? '' + const db = relationUnavailable(workspace, 'databases', ns, 'Databases') + ? null + : databaseForDeclaration(resource, workspaceList(workspace, 'databases')) + return ( + + {dbname} + {db && ( + + {' · declared by Database '} + + + )} + + ) +} + +function LinkList({ items, kind, onNavigate }: { items: any[]; kind: string; onNavigate?: CNPGNavigate }) { + return ( + + {items.map((o) => ( + + ))} + + ) +} + +export function CNPGDatabaseSummary({ resource, workspace, onNavigate }: SummaryProps) { + const ns = resource?.metadata?.namespace ?? '' + const cluster = targetCluster(resource, clustersIn(workspace)) + const missingRole = missingManagedRole(resource, cluster) + const pubsUnavailable = relationUnavailable(workspace, 'publications', ns, 'Publications') + const subsUnavailable = relationUnavailable(workspace, 'subscriptions', ns, 'Subscriptions') + const related = replicationForDatabase(resource, workspaceList(workspace, 'publications'), workspaceList(workspace, 'subscriptions')) + + return ( + + + + Declared + + + {resource?.spec?.name ? {resource.spec.name} : } + + + {resource?.spec?.owner ? {resource.spec.owner} : } + + {resource?.spec?.ensure ?? 'present'} + + + + + “{missingRole}” is not among {cluster?.metadata?.name}'s managed roles + + ) + } + /> + + Source and target + + + + + + + + + {pubsUnavailable ? ( + + ) : related.publications.length === 0 ? ( + None on this database + ) : ( + + )} + + + {subsUnavailable ? ( + + ) : related.subscriptions.length === 0 ? ( + None on this database + ) : ( + + )} + + + + ) +} + +function publicationTargets(resource: any): ReactNode { + const target = resource?.spec?.target + if (target?.allTables === true) return 'All tables' + const objects = Array.isArray(target?.objects) ? target.objects : [] + if (objects.length === 0) return + const labels = objects.map((o: any) => { + if (o?.tablesInSchema) return `All tables in schema ${o.tablesInSchema}` + const t = o?.table + if (t?.name) { + const name = t.schema ? `${t.schema}.${t.name}` : t.name + return Array.isArray(t.columns) && t.columns.length > 0 ? `${name} (${t.columns.join(', ')})` : name + } + return 'Unrecognized entry' + }) + return ( +
    + {labels.map((l: string, i: number) => ( +
  • {l}
  • + ))} +
+ ) +} + +export function CNPGPublicationSummary({ resource, workspace, onNavigate }: SummaryProps) { + return ( + + + + Declared + + + {resource?.spec?.name ? {resource.spec.name} : } + + + + + + + + {publicationTargets(resource)} + + + + + + + + + ) +} + +export function CNPGSubscriptionSummary({ resource, workspace, onNavigate }: SummaryProps) { + const pub = resource?.spec?.publicationName + const ext = resource?.spec?.externalClusterName + return ( + + + + Declared + + + {resource?.spec?.name ? {resource.spec.name} : } + + + + + + + + + {pub ? ( + + Publication {pub} + {ext ? ( + + {' on external cluster '} + {ext} + + ) : null} + + ) : ( + + )} + + + + + + + + + + ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx new file mode 100644 index 000000000..7b89de5da --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx @@ -0,0 +1,76 @@ +import { CNPG_GROUP, getCNPGImageCatalogEntries } from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' +import { NotReported, Note, ObjectProblems, SummaryShell } from './CNPGSharedSummary' +import { clustersIn, clustersUsingCatalog, refOf, relationUnavailable } from './relations' + +export function CNPGImageCatalogSummary({ + resource, + workspace, + onNavigate, +}: { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +}) { + const clusterScoped = resource?.kind === 'ClusterImageCatalog' + const ns = resource?.metadata?.namespace ?? '' + const entries = getCNPGImageCatalogEntries(resource) + const majors = new Set(entries.map((e) => e.major)) + const unavailable = relationUnavailable(workspace, 'clusters', clusterScoped ? undefined : ns, 'Clusters') + const users = unavailable ? [] : clustersUsingCatalog(resource, clustersIn(workspace)) + + return ( + + + + Images + {entries.length === 0 ? ( +
+ +
+ ) : ( + + {entries.map((e) => ( + + {e.image} + + ))} + + )} + + Used by + {unavailable ? ( +
+ +
+ ) : users.length === 0 ? ( +
No visible cluster uses this catalog
+ ) : ( + + {users.map((u) => ( + + {clusterScoped ? `${u.cluster.metadata?.namespace}/${u.cluster.metadata?.name}` : u.cluster.metadata?.name} + + } + > + {u.major === null ? ( + + ) : majors.has(u.major) ? ( + `Requests PostgreSQL ${u.major}` + ) : ( + Requests PostgreSQL {u.major} · not in this catalog + )} + + ))} + + )} + {clusterScoped && !unavailable && ( + Among clusters you can see; clusters in namespaces you cannot read are not listed + )} +
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx new file mode 100644 index 000000000..bbda3bb7d --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx @@ -0,0 +1,174 @@ +import { + CNPG_BARMAN_OBJECTSTORE_GROUP, + CNPG_GROUP, + getCNPGObjectStoreCredentialSecret, + getCNPGObjectStoreDestination, + getCNPGObjectStoreProvider, + getCNPGObjectStoreRecoveryWindows, + getCNPGObjectStoreRetention, +} from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' +import { NotReported, Note, ObjectProblems, SummaryShell, TimeAgo } from './CNPGSharedSummary' +import { clustersIn, inferredObjectStoreHealth, refOf, relationUnavailable, usersOfObjectStore } from './relations' + +function utc(at: string | undefined): string { + if (!at || !Number.isFinite(Date.parse(at))) return 'unknown' + return new Date(at).toUTCString().replace(' GMT', ' UTC') +} + +export function CNPGObjectStoreSummary({ + resource, + workspace, + onNavigate, +}: { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +}) { + const ns = resource?.metadata?.namespace ?? '' + const clustersUnavailable = relationUnavailable(workspace, 'clusters', ns, 'Clusters') + const users = clustersUnavailable ? [] : usersOfObjectStore(resource, clustersIn(workspace)) + const health = inferredObjectStoreHealth(resource, users) + const clusterForServer = new Map(users.map((u) => [u.serverName, u.cluster?.metadata?.name as string])) + const windows = getCNPGObjectStoreRecoveryWindows(resource) + const destination = getCNPGObjectStoreDestination(resource) + const provider = getCNPGObjectStoreProvider(resource) + const secret = getCNPGObjectStoreCredentialSecret(resource) + const retention = getCNPGObjectStoreRetention(resource) + const clusterWord = users.length === 1 ? "1 cluster's" : `${users.length} clusters'` + + return ( + + + + Upload health + {clustersUnavailable ? ( +
+ +
+ ) : ( + <> +
+ +
+ {users.length > 0 && ( + Inferred from {clusterWord} WAL archiving and backup results — ObjectStore has no health status + )} + {health.evidence.length > 0 && ( +
+ + {health.evidence.map((e) => ( + }> +
+ +
+ {e.window ? ( + <> + Last backup success + {e.window.lastFailedBackupTime && ( + + {' · '}last failure + + )} + + ) : ( + + )} +
+
+
+ ))} +
+
+ )} + + )} + + Recovery window + {windows.length === 0 ? ( +
+ +
+ ) : ( + + {windows.map((w) => { + const cluster = clusterForServer.get(w.server) + return ( + + {w.server} + + ) : ( + {w.server} + ) + } + > +
+ {w.firstRecoverabilityPoint || w.lastSuccessfulBackupTime ? ( + + {utc(w.firstRecoverabilityPoint)} → {utc(w.lastSuccessfulBackupTime)} + + ) : ( + + )} + {w.lastFailedBackupTime && ( +
+ Last failed backup +
+ )} + {w.failingSinceLastSuccess && ( + The window is still restorable up to the last successful backup; it stops advancing while uploads fail. + )} +
+
+ ) + })} +
+ )} + + Destination + + {destination !== '-' ? {destination} : } + {provider ?? } + + {secret ? ( + + Secret + + ) : ( + + )} + + {retention ?? } + + + Used by + {clustersUnavailable ? ( +
+ +
+ ) : users.length === 0 ? ( +
+ No visible cluster uses this store + {workspace?.coverage?.clusters?.state === 'partial' && ( + Clusters in namespaces you cannot read are not checked. + )} +
+ ) : ( +
+ {users.map((u) => ( + + ))} +
+ )} +
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx new file mode 100644 index 000000000..2e6ba3cd2 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx @@ -0,0 +1,193 @@ +import { describe, expect, it } from 'vitest' +import { renderToString } from 'react-dom/server' +import { CNPGBackupSummary, CNPGScheduledBackupSummary } from './CNPGBackupSummary' +import { CNPGObjectStoreSummary } from './CNPGObjectStoreSummary' +import { CNPGDatabaseSummary } from './CNPGDeclarativeSummary' +import { CNPGPoolerSummary } from './CNPGPoolerSummary' +import { CNPGImageCatalogSummary } from './CNPGImageCatalogSummary' +import { CNPG_WORKSPACE_KEYS, type CNPGWorkspaceKey, type CNPGWorkspaceResponse } from './workspace' + +const PG = 'postgresql.cnpg.io/v1' +const PLUGIN = 'barman-cloud.cloudnative-pg.io' +const nav = () => {} + +function ws(objects: Partial>, over: Partial = {}): CNPGWorkspaceResponse { + const coverage: CNPGWorkspaceResponse['coverage'] = {} + for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } + return { + installed: true, + context: 'c', + namespaces: ['pg'], + coverage: { ...coverage, ...(over.coverage ?? {}) }, + objects, + issues: over.issues ?? [], + audit: [], + backupsOmitted: over.backupsOmitted ?? 0, + } +} + +function text(html: string): string { + return html.replace(/<[^>]+>/g, '').replace(/'/g, "'").replace(/"/g, '"').replace(/&/g, '&') +} + +const mainCluster = { + apiVersion: PG, + kind: 'Cluster', + metadata: { name: 'main', namespace: 'pg' }, + spec: { plugins: [{ name: PLUGIN, parameters: { barmanObjectName: 'store' } }], managed: { roles: [{ name: 'reporting' }] } }, + status: { conditions: [{ type: 'ContinuousArchiving', status: 'False', message: 'upload failed' }] }, +} + +describe('CNPGBackupSummary', () => { + const b = { + apiVersion: PG, + kind: 'Backup', + metadata: { name: 'main-20260901', namespace: 'pg', labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, + spec: { cluster: { name: 'main' }, method: 'plugin', pluginConfiguration: { name: PLUGIN } }, + status: { + phase: 'failed', + startedAt: '2026-09-01T00:00:00Z', + stoppedAt: '2026-09-01T00:02:30Z', + instanceID: { podName: 'main-2' }, + error: 'can not upload', + }, + } + + it('shows the outcome and relationships from the same status fields', () => { + const html = renderToString() + const t = text(html) + expect(t).toContain('Failed') + expect(t).toContain('took 2m 30s') + expect(t).toContain('main-2') + expect(t).toContain('can not upload') + expect(t).toContain('ScheduledBackup nightly') + expect(t).toContain('ObjectStore store') + expect(t).not.toContain('On demand') + }) + + it('shows its own issues on top', () => { + const issues = [ + { id: 'i1', severity: 'critical' as const, kind: 'Backup', group: 'postgresql.cnpg.io', namespace: 'pg', name: 'main-20260901', reason: 'CNPGBackupFailed', message: 'Backup failed' }, + { id: 'i2', severity: 'critical' as const, kind: 'Backup', group: 'velero.io', namespace: 'pg', name: 'main-20260901', reason: 'VeleroBackupFailed', message: 'Velero backup failed' }, + ] + const t = text(renderToString()) + expect(t).toContain('Backup failed') + expect(t).not.toContain('Velero backup failed') + }) +}) + +describe('CNPGScheduledBackupSummary', () => { + it('lists owned runs and notes omitted history', () => { + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' }, spec: { cluster: { name: 'main' }, schedule: '0 0 0 * * *' } } + const run = { apiVersion: PG, kind: 'Backup', metadata: { name: 'run-1', namespace: 'pg', labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, status: { phase: 'completed', startedAt: '2026-09-01T00:00:00Z' } } + const t = text(renderToString()) + expect(t).toContain('run-1') + expect(t).toContain('Completed') + expect(t).toContain('Backups older than 7 days are not listed') + }) + + it('says when Backups are not readable instead of listing none', () => { + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' }, spec: { cluster: { name: 'main' } } } + const t = text(renderToString()) + expect(t).toContain('No access to Backups') + expect(t).not.toContain('No Backups from this schedule') + }) +}) + +describe('CNPGObjectStoreSummary', () => { + const store = { + apiVersion: 'barmancloud.cnpg.io/v1', + kind: 'ObjectStore', + metadata: { name: 'store', namespace: 'pg' }, + spec: { + configuration: { + destinationPath: 's3://bucket/pg', + s3Credentials: { accessKeyId: { name: 's3-creds', key: 'ACCESS_KEY_ID' }, secretAccessKey: { name: 's3-creds', key: 'SECRET' } }, + }, + retentionPolicy: '30d', + }, + status: { + serverRecoveryWindow: { + main: { firstRecoverabilityPoint: '2026-09-01T00:00:00Z', lastSuccessfulBackupTime: '2026-09-02T00:00:00Z', lastFailedBackupTime: '2026-09-03T00:00:00Z' }, + }, + }, + } + + it('labels upload health as inferred and explains the stalled window', () => { + const t = text(renderToString()) + expect(t).toContain('inferred') + expect(t).toContain("Inferred from 1 cluster's WAL archiving and backup results — ObjectStore has no health status") + expect(t).toContain('Uploads failing') + expect(t).toContain('The window is still restorable up to the last successful backup; it stops advancing while uploads fail.') + expect(t).toContain('s3-creds') + expect(t).toContain('s3://bucket/pg') + }) + + it('never renders credential keys or Secret contents', () => { + const html = renderToString() + expect(html).not.toContain('ACCESS_KEY_ID') + expect(html).not.toContain('SECRET') + }) + + it('says when no visible cluster uses the store', () => { + const t = text(renderToString()) + expect(t).toContain('No visible cluster uses this store') + expect(t).not.toContain('Inferred from') + }) +}) + +describe('CNPGDatabaseSummary', () => { + const db = (status: any) => ({ + apiVersion: PG, + kind: 'Database', + metadata: { name: 'app-db', namespace: 'pg', generation: 2 }, + spec: { cluster: { name: 'main' }, name: 'app', owner: 'app', databaseReclaimPolicy: 'delete' }, + status, + }) + + it('reports an unreconciled database as pending, not failed', () => { + const t = text(renderToString()) + expect(t).toContain('Pending') + expect(t).not.toContain('Not applied') + expect(t).toContain('drops it from PostgreSQL') + expect(t).toContain('Applied directly (no GitOps owner label)') + }) + + it('reports a failure and the missing managed role beside it', () => { + const t = text( + renderToString( + , + ), + ) + expect(t).toContain('Not applied') + expect(t).toContain('“app” is not among main\'s managed roles') + }) +}) + +describe('CNPGPoolerSummary', () => { + it('reports unknown scheduled count and unmeasured pressure', () => { + const pooler = { apiVersion: PG, kind: 'Pooler', metadata: { name: 'main-rw', namespace: 'pg' }, spec: { cluster: { name: 'main' }, type: 'rw', instances: 2 } } + const t = text(renderToString()) + expect(t).toContain('Scheduled count not reported') + expect(t).toContain('Not measured') + expect(t).toContain('main-rw') + }) +}) + +describe('CNPGImageCatalogSummary', () => { + it('lists users and flags a major the catalog lacks', () => { + const catalog = { apiVersion: PG, kind: 'ClusterImageCatalog', metadata: { name: 'pg' }, spec: { images: [{ major: 16, image: 'ghcr.io/cnpg/postgresql:16' }] } } + const clusters = [ + { apiVersion: PG, kind: 'Cluster', metadata: { name: 'a', namespace: 'x' }, spec: { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 16 } } }, + { apiVersion: PG, kind: 'Cluster', metadata: { name: 'b', namespace: 'y' }, spec: { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 17 } } }, + ] + const t = text(renderToString()) + expect(t).toContain('x/a') + expect(t).toContain('Requests PostgreSQL 17 · not in this catalog') + expect(t).toContain('clusters in namespaces you cannot read are not listed') + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx new file mode 100644 index 000000000..9418de062 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx @@ -0,0 +1,70 @@ +import { getCNPGPoolerDeploymentName, getCNPGPoolerMode, getCNPGPoolerStatus, isCNPGPoolerPaused } from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceResponse } from './workspace' +import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, type CNPGNavigate } from './primitives' +import { ClusterLink, NotReported, Note, ObjectProblems, PhaseBadge, SummaryShell } from './CNPGSharedSummary' +import { refOf } from './relations' + +const TYPE_LABEL: Record = { + rw: 'rw · routes to the primary', + ro: 'ro · routes to replicas', + r: 'r · routes to any instance', +} + +export function CNPGPoolerSummary({ + resource, + workspace, + onNavigate, +}: { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +}) { + const ns = resource?.metadata?.namespace ?? '' + const type = resource?.spec?.type + const desired = resource?.spec?.instances + const scheduled = resource?.status?.instances + const deployment = getCNPGPoolerDeploymentName(resource) + + return ( + + + + State + + + + {isCNPGPoolerPaused(resource) && PgBouncer is paused: it holds client connections instead of serving them.} + + + + {typeof scheduled === 'number' ? `${scheduled} scheduled` : } + + {' · '} + {typeof desired === 'number' ? `${desired} desired` : 'desired not set'} + + + The Pooler counts scheduled pods, not ready ones; readiness is on its Deployment. + + + {deployment ? ( + + ) : ( + + )} + + + + + + + Routing + + + + + {type ? TYPE_LABEL[type] ?? type : } + {getCNPGPoolerMode(resource)} + + + ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx new file mode 100644 index 000000000..301ee5ad2 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx @@ -0,0 +1,88 @@ +import type { ReactNode } from 'react' +import { Badge } from '../ui/Badge' +import type { StatusBadge as StatusBadgeValue } from '../resources/resource-utils' +import { CNPG_GROUP } from '../resources/resource-utils-cnpg' +import type { CNPGWorkspaceIssue, CNPGWorkspaceResponse } from './workspace' +import { FactValue, ProblemCallout, RefLink, type CNPGNavigate } from './primitives' +import { clustersIn, healthSeverity, problemsForObject, relationUnavailable, targetCluster, type CNPGObjectRef } from './relations' + +const MAX_PROBLEMS = 3 + +export function SummaryShell({ children }: { children: ReactNode }) { + return
{children}
+} + +/** The object's own Radar issues, most severe first. */ +export function ObjectProblems({ + issues, + subject, + onNavigate, +}: { + issues: CNPGWorkspaceIssue[] | undefined + subject: CNPGObjectRef + onNavigate?: CNPGNavigate +}) { + const problems = problemsForObject(issues, subject) + if (problems.length === 0) return null + const shown = problems.slice(0, MAX_PROBLEMS) + const rest = problems.length - shown.length + return ( +
+ {shown.map((p, i) => ( + 0 ? +{rest} more : null} + /> + ))} +
+ ) +} + +export function PhaseBadge({ status }: { status: StatusBadgeValue }) { + return ( + + {status.text} + + ) +} + +export function NotReported({ text = 'Not reported' }: { text?: string }) { + return {text} +} + +/** A timestamp as an age, with the absolute time on hover. */ +export function TimeAgo({ at, missing }: { at: unknown; missing?: string }) { + if (typeof at !== 'string' || !at || !Number.isFinite(Date.parse(at))) return + return +} + +export function Note({ children }: { children: ReactNode }) { + return
{children}
+} + +/** The Cluster an object declares itself against, linked. */ +export function ClusterLink({ + resource, + workspace, + onNavigate, +}: { + resource: any + workspace: CNPGWorkspaceResponse | null + onNavigate?: CNPGNavigate +}) { + const name = resource?.spec?.cluster?.name + if (!name) return + const ns = resource?.metadata?.namespace ?? '' + const visible = !!targetCluster(resource, clustersIn(workspace)) + return ( + + + {workspace && !visible && !relationUnavailable(workspace, 'clusters', ns, 'Clusters') && ( + · not found in this namespace + )} + + ) +} diff --git a/packages/k8s-ui/src/components/cnpg/index.ts b/packages/k8s-ui/src/components/cnpg/index.ts index adbdba2df..026c0b142 100644 --- a/packages/k8s-ui/src/components/cnpg/index.ts +++ b/packages/k8s-ui/src/components/cnpg/index.ts @@ -1,3 +1,8 @@ export * from './workspace' export * from './primitives' export * from './CNPGClusterSummary' +export * from './CNPGBackupSummary' +export * from './CNPGObjectStoreSummary' +export * from './CNPGDeclarativeSummary' +export * from './CNPGPoolerSummary' +export * from './CNPGImageCatalogSummary' diff --git a/packages/k8s-ui/src/components/cnpg/primitives.tsx b/packages/k8s-ui/src/components/cnpg/primitives.tsx index 711e5680c..7020a4da8 100644 --- a/packages/k8s-ui/src/components/cnpg/primitives.tsx +++ b/packages/k8s-ui/src/components/cnpg/primitives.tsx @@ -102,13 +102,16 @@ export function ProblemCallout({ more, onNavigate, action, + subjectIsSelf, }: { problem: CNPGProblem more?: ReactNode onNavigate?: CNPGNavigate action?: ReactNode + /** The callout sits on the subject's own page, so linking to it would loop. */ + subjectIsSelf?: boolean }) { - const aboutChild = problem.subject.kind !== 'Cluster' + const aboutChild = !subjectIsSelf && problem.subject.kind !== 'Cluster' return (
diff --git a/packages/k8s-ui/src/components/cnpg/relations.test.ts b/packages/k8s-ui/src/components/cnpg/relations.test.ts new file mode 100644 index 000000000..5d136500e --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/relations.test.ts @@ -0,0 +1,289 @@ +import { describe, expect, it } from 'vitest' +import { + appliedFact, + backupDestination, + backupsForScheduledBackup, + clustersUsingCatalog, + databaseForDeclaration, + gitopsSourceOf, + inferredObjectStoreHealth, + issuesForObject, + missingManagedRole, + objectStoreForBackup, + relationUnavailable, + replicationForDatabase, + scheduledBackupOf, + usersOfObjectStore, +} from './relations' +import { CNPG_WORKSPACE_KEYS, type CNPGWorkspaceIssue, type CNPGWorkspaceResponse } from './workspace' + +const PG = 'postgresql.cnpg.io/v1' +const BARMAN = 'barmancloud.cnpg.io/v1' +const VELERO = 'velero.io/v1' +const PLUGIN = 'barman-cloud.cloudnative-pg.io' + +function cluster(name: string, ns = 'pg', spec: any = {}, status: any = {}): any { + return { apiVersion: PG, kind: 'Cluster', metadata: { name, namespace: ns }, spec, status } +} + +function pluginCluster(name: string, store: string, serverName?: string, conditions?: any[]): any { + return cluster( + name, + 'pg', + { plugins: [{ name: PLUGIN, isWALArchiver: true, parameters: { barmanObjectName: store, ...(serverName ? { serverName } : {}) } }] }, + conditions ? { conditions } : {}, + ) +} + +function backup(name: string, extra: any = {}): any { + return { + apiVersion: PG, + kind: 'Backup', + metadata: { name, namespace: 'pg', ...(extra.metadata ?? {}) }, + spec: { cluster: { name: 'main' }, ...(extra.spec ?? {}) }, + status: extra.status ?? {}, + } +} + +describe('scheduledBackupOf / backupsForScheduledBackup', () => { + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' } } + + it('reads the owner reference', () => { + const b = backup('b1', { metadata: { ownerReferences: [{ apiVersion: PG, kind: 'ScheduledBackup', name: 'nightly' }] } }) + expect(scheduledBackupOf(b)).toBe('nightly') + }) + + it('falls back to the operator label when the owner is the Cluster or unset', () => { + const b = backup('b1', { + metadata: { + labels: { 'cnpg.io/scheduled-backup': 'nightly' }, + ownerReferences: [{ apiVersion: PG, kind: 'Cluster', name: 'main' }], + }, + }) + expect(scheduledBackupOf(b)).toBe('nightly') + }) + + it('returns null for an on-demand backup', () => { + expect(scheduledBackupOf(backup('b1'))).toBeNull() + }) + + it('ignores an owner reference from another API group', () => { + const b = backup('b1', { metadata: { ownerReferences: [{ apiVersion: 'example.com/v1', kind: 'ScheduledBackup', name: 'nightly' }] } }) + expect(scheduledBackupOf(b)).toBeNull() + }) + + it('lists owned backups newest first and never a Velero Backup', () => { + const owned = (name: string, startedAt: string, apiVersion = PG) => ({ + ...backup(name, { metadata: { labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, status: { startedAt } }), + apiVersion, + }) + const list = [ + owned('old', '2026-09-01T00:00:00Z'), + owned('new', '2026-09-02T00:00:00Z'), + owned('velero', '2026-09-03T00:00:00Z', VELERO), + backup('other'), + { ...owned('elsewhere', '2026-09-03T00:00:00Z'), metadata: { name: 'elsewhere', namespace: 'x', labels: { 'cnpg.io/scheduled-backup': 'nightly' } } }, + ] + expect(backupsForScheduledBackup(sched, list).map((b) => b.metadata.name)).toEqual(['new', 'old']) + }) +}) + +describe('objectStoreForBackup / backupDestination', () => { + const clusters = [pluginCluster('main', 'store-a')] + + it('prefers the backup plugin parameters', () => { + const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN, parameters: { barmanObjectName: 'store-b' } } } }) + expect(objectStoreForBackup(b, clusters)).toBe('store-b') + }) + + it('falls back to the target cluster plugin', () => { + const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN } } }) + expect(objectStoreForBackup(b, clusters)).toBe('store-a') + expect(backupDestination(b, clusters)).toEqual({ type: 'objectStore', name: 'store-a' }) + }) + + it('does not attribute another plugin to the barman store', () => { + const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: 'other.example.com' } } }) + expect(objectStoreForBackup(b, clusters)).toBeNull() + }) + + it('reports in-tree paths and volume snapshots', () => { + expect(backupDestination(backup('b', { status: { method: 'barmanObjectStore', destinationPath: 's3://x' } }), clusters)).toEqual({ type: 'path', path: 's3://x' }) + expect(backupDestination(backup('b', { spec: { method: 'volumeSnapshot' } }), clusters)).toEqual({ type: 'volumeSnapshot' }) + expect(backupDestination(backup('b'), clusters)).toEqual({ type: 'unknown' }) + }) +}) + +describe('ObjectStore users and inferred health', () => { + const store = { + apiVersion: BARMAN, + kind: 'ObjectStore', + metadata: { name: 'store', namespace: 'pg' }, + status: { + serverRecoveryWindow: { + 'srv-a': { firstRecoverabilityPoint: '2026-09-01T00:00:00Z', lastSuccessfulBackupTime: '2026-09-02T00:00:00Z' }, + b: { lastSuccessfulBackupTime: '2026-09-02T00:00:00Z', lastFailedBackupTime: '2026-09-03T00:00:00Z' }, + }, + }, + } + const ok = [{ type: 'ContinuousArchiving', status: 'True' }] + + it('finds users in the same namespace, keyed by serverName', () => { + const clusters = [ + pluginCluster('a', 'store', 'srv-a', ok), + pluginCluster('b', 'store'), + pluginCluster('c', 'other'), + { ...pluginCluster('d', 'store'), metadata: { name: 'd', namespace: 'elsewhere' } }, + { ...pluginCluster('e', 'store'), apiVersion: 'cluster.x-k8s.io/v1beta1' }, + ] + const users = usersOfObjectStore(store, clusters) + expect(users.map((u) => [u.cluster.metadata.name, u.serverName])).toEqual([ + ['a', 'srv-a'], + ['b', 'b'], + ]) + }) + + it('reports failing when any user has a failure newer than its last success', () => { + const users = usersOfObjectStore(store, [pluginCluster('a', 'store', 'srv-a', ok), pluginCluster('b', 'store', undefined, ok)]) + const h = inferredObjectStoreHealth(store, users) + expect(h.summary.tone).toBe('unhealthy') + expect(h.summary.text).toBe('Uploads failing for 1 of 2 clusters') + expect(h.summary.at).toBe('2026-09-03T00:00:00Z') + }) + + it('reports failing from a False ContinuousArchiving condition', () => { + const users = usersOfObjectStore(store, [pluginCluster('a', 'store', 'srv-a', [{ type: 'ContinuousArchiving', status: 'False', message: 'boom' }])]) + expect(inferredObjectStoreHealth(store, users).summary).toMatchObject({ text: 'Uploads failing', tone: 'unhealthy' }) + }) + + it('is healthy only when every user reports archiving', () => { + const users = usersOfObjectStore(store, [pluginCluster('a', 'store', 'srv-a', ok)]) + expect(inferredObjectStoreHealth(store, users).summary.tone).toBe('healthy') + const unreported = usersOfObjectStore(store, [pluginCluster('a', 'store', 'srv-a')]) + expect(inferredObjectStoreHealth(store, unreported).summary).toMatchObject({ text: 'No failures reported', tone: 'unknown' }) + }) + + it('says so when nothing uses the store', () => { + expect(inferredObjectStoreHealth(store, []).summary).toMatchObject({ text: 'No visible cluster uses this store', tone: 'unknown' }) + }) +}) + +describe('clustersUsingCatalog', () => { + const catalog = { apiVersion: PG, kind: 'ImageCatalog', metadata: { name: 'pg', namespace: 'pg' } } + const clusterCatalog = { apiVersion: PG, kind: 'ClusterImageCatalog', metadata: { name: 'pg' } } + + it('treats an omitted ref kind as ImageCatalog and stays in the namespace', () => { + const clusters = [ + cluster('a', 'pg', { imageCatalogRef: { name: 'pg', major: 16 } }), + cluster('b', 'pg', { imageCatalogRef: { kind: 'ImageCatalog', name: 'pg', major: 17 } }), + cluster('c', 'other', { imageCatalogRef: { name: 'pg', major: 16 } }), + cluster('d', 'pg', { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 16 } }), + ] + expect(clustersUsingCatalog(catalog, clusters).map((u) => [u.cluster.metadata.name, u.major])).toEqual([ + ['a', 16], + ['b', 17], + ]) + }) + + it('matches ClusterImageCatalog refs from any namespace, never an omitted kind', () => { + const clusters = [ + cluster('a', 'x', { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 16 } }), + cluster('b', 'y', { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg' } }), + cluster('c', 'pg', { imageCatalogRef: { name: 'pg', major: 16 } }), + ] + expect(clustersUsingCatalog(clusterCatalog, clusters).map((u) => [u.cluster.metadata.name, u.major])).toEqual([ + ['a', 16], + ['b', null], + ]) + }) +}) + +describe('issuesForObject', () => { + const issue = (over: Partial): CNPGWorkspaceIssue => ({ + id: Math.random().toString(), + severity: 'warning', + kind: 'Backup', + group: 'postgresql.cnpg.io', + namespace: 'pg', + name: 'b1', + reason: 'CNPGBackupFailed', + ...over, + }) + + it('matches kind, group, namespace and name, critical first', () => { + const issues = [ + issue({ id: 'w' }), + issue({ id: 'c', severity: 'critical' }), + issue({ id: 'velero', group: 'velero.io' }), + issue({ id: 'ns', namespace: 'other' }), + issue({ id: 'name', name: 'b2' }), + issue({ id: 'kind', kind: 'ScheduledBackup' }), + ] + const got = issuesForObject(issues, { kind: 'Backup', group: 'postgresql.cnpg.io', namespace: 'pg', name: 'b1' }) + expect(got.map((i) => i.id)).toEqual(['c', 'w']) + }) +}) + +describe('declarations', () => { + it('distinguishes pending from failed', () => { + expect(appliedFact({ status: {} }).tone).toBe('unknown') + expect(appliedFact({ status: { applied: false } }).text).toBe('Not applied') + expect(appliedFact({ status: { applied: true } }).text).toBe('Applied') + }) + + it('names a missing role only when it is absent from managed roles', () => { + const db = { status: { applied: false, message: 'role "app" does not exist' } } + expect(missingManagedRole(db, cluster('main', 'pg', { managed: { roles: [{ name: 'other' }] } }))).toBe('app') + expect(missingManagedRole(db, cluster('main', 'pg', { managed: { roles: [{ name: 'app' }] } }))).toBeNull() + expect(missingManagedRole(db, null)).toBeNull() + expect(missingManagedRole({ status: { message: 'connection refused' } }, cluster('main'))).toBeNull() + }) + + it('reads the GitOps owner labels', () => { + expect(gitopsSourceOf({ metadata: { labels: { 'argocd.argoproj.io/instance': 'app' } } })).toEqual({ tool: 'argocd', name: 'app' }) + expect(gitopsSourceOf({ metadata: { labels: { 'kustomize.toolkit.fluxcd.io/name': 'k', 'kustomize.toolkit.fluxcd.io/namespace': 'flux' } } })).toEqual({ + tool: 'flux', + name: 'k', + namespace: 'flux', + }) + expect(gitopsSourceOf({ metadata: {} })).toBeNull() + }) + + const decl = (kind: string, name: string, clusterName: string, dbname: string, ns = 'pg') => ({ + apiVersion: PG, + kind, + metadata: { name, namespace: ns }, + spec: { cluster: { name: clusterName }, dbname }, + }) + const database = { apiVersion: PG, kind: 'Database', metadata: { name: 'app-db', namespace: 'pg' }, spec: { cluster: { name: 'main' }, name: 'app' } } + + it('finds publications and subscriptions on the same cluster and database', () => { + const pubs = [decl('Publication', 'p1', 'main', 'app'), decl('Publication', 'p2', 'main', 'other'), decl('Publication', 'p3', 'other', 'app')] + const subs = [decl('Subscription', 's1', 'main', 'app'), decl('Subscription', 's2', 'main', 'app', 'x')] + const r = replicationForDatabase(database, pubs, subs) + expect(r.publications.map((p) => p.metadata.name)).toEqual(['p1']) + expect(r.subscriptions.map((s) => s.metadata.name)).toEqual(['s1']) + }) + + it('finds the Database declaring a publication database', () => { + expect(databaseForDeclaration(decl('Publication', 'p1', 'main', 'app'), [database])?.metadata.name).toBe('app-db') + expect(databaseForDeclaration(decl('Publication', 'p1', 'other', 'app'), [database])).toBeNull() + }) +}) + +describe('relationUnavailable', () => { + const ws = (state: any, deniedNamespaces?: string[]): CNPGWorkspaceResponse => { + const coverage: CNPGWorkspaceResponse['coverage'] = {} + for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } + coverage.backups = { state, deniedNamespaces } + return { installed: true, context: 'c', namespaces: null, coverage, objects: {}, issues: [], audit: [], backupsOmitted: 0 } + } + + it('answers from coverage', () => { + expect(relationUnavailable(ws('full'), 'backups', 'pg', 'Backups')).toBeNull() + expect(relationUnavailable(ws('partial', ['pg']), 'backups', 'pg', 'Backups')).toBe('No access to Backups') + expect(relationUnavailable(ws('partial', ['other']), 'backups', 'pg', 'Backups')).toBeNull() + expect(relationUnavailable(ws('denied'), 'backups', 'pg', 'Backups')).toBe('No access to Backups') + expect(relationUnavailable(null, 'backups', 'pg', 'Backups')).toBe('Backups could not be read') + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/relations.ts b/packages/k8s-ui/src/components/cnpg/relations.ts new file mode 100644 index 000000000..cc840ddbd --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/relations.ts @@ -0,0 +1,398 @@ +// Pure relationship lookups between CloudNativePG objects in the workspace +// payload. Each helper answers only from what the objects record; a relation +// that cannot be established returns null or an empty list, never a guess. + +import type { BadgeSeverity } from '../ui/Badge' +import type { HealthLevel } from '../resources/resource-utils' +import { + CNPG_BARMAN_PLUGIN_NAME, + CNPG_GROUP, + getCNPGClusterBarmanPlugin, + getCNPGObjectStoreRecoveryWindows, + isApiGroup, + type CNPGObjectStoreRecoveryWindow, +} from '../resources/resource-utils-cnpg' +import { + cnpgIssueCategory, + coverageReadable, + type CNPGFact, + type CNPGProblem, + type CNPGWorkspaceIssue, + type CNPGWorkspaceKey, + type CNPGWorkspaceResponse, +} from './workspace' + +export interface CNPGObjectRef { + kind: string + group: string + namespace: string + name: string +} + +const SCHEDULED_BACKUP_LABEL = 'cnpg.io/scheduled-backup' + +function nsOf(obj: any): string { + return obj?.metadata?.namespace ?? '' +} + +function nameOf(obj: any): string { + return obj?.metadata?.name ?? '' +} + +function specCluster(obj: any): string | undefined { + const n = obj?.spec?.cluster?.name + return typeof n === 'string' && n ? n : undefined +} + +function parseTime(t: unknown): number { + const ms = typeof t === 'string' ? Date.parse(t) : NaN + return Number.isFinite(ms) ? ms : 0 +} + +/** Kind and API group both match. Kind alone is not identity: Velero also ships `Backup`. */ +export function isCNPGKind(obj: any, kind: string, group: string = CNPG_GROUP): boolean { + return obj?.kind === kind && isApiGroup(obj?.apiVersion, group) +} + +export function refOf(obj: any, kind: string, group: string = CNPG_GROUP): CNPGObjectRef { + return { kind, group, namespace: nsOf(obj), name: nameOf(obj) } +} + +export function healthSeverity(level: HealthLevel): BadgeSeverity { + switch (level) { + case 'healthy': + return 'success' + case 'unhealthy': + return 'error' + case 'alert': + return 'alert' + case 'degraded': + return 'warning' + default: + return 'neutral' + } +} + +// --------------------------------------------------------------------------- +// Workspace access +// --------------------------------------------------------------------------- + +export function workspaceList(ws: CNPGWorkspaceResponse | null | undefined, key: CNPGWorkspaceKey): any[] { + return ws?.objects?.[key] ?? [] +} + +/** + * Why the workspace cannot answer for `key` in `namespace`, or null when it can. + * A null workspace means the aggregate could not be read at all. + */ +export function relationUnavailable( + ws: CNPGWorkspaceResponse | null | undefined, + key: CNPGWorkspaceKey, + namespace: string | undefined, + what: string, +): string | null { + if (!ws) return `${what} could not be read` + const cov = ws.coverage?.[key] ?? { state: 'notInstalled' as const } + if (coverageReadable(cov, namespace)) return null + switch (cov.state) { + case 'denied': + case 'partial': + return `No access to ${what}` + case 'syncing': + return 'Loading…' + case 'error': + return `Could not read ${what}` + default: + return `${what} are not installed` + } +} + +export function clustersIn(ws: CNPGWorkspaceResponse | null | undefined): any[] { + return workspaceList(ws, 'clusters').filter((c) => isCNPGKind(c, 'Cluster')) +} + +/** The Cluster an object declares itself against (`spec.cluster.name`), if visible. */ +export function targetCluster(obj: any, clusters: any[]): any | null { + const name = specCluster(obj) + if (!name) return null + return clusters.find((c) => isCNPGKind(c, 'Cluster') && nsOf(c) === nsOf(obj) && nameOf(c) === name) ?? null +} + +// --------------------------------------------------------------------------- +// Issues +// --------------------------------------------------------------------------- + +export function issuesForObject(issues: CNPGWorkspaceIssue[] | undefined, ref: CNPGObjectRef): CNPGWorkspaceIssue[] { + const rank = { critical: 0, warning: 1 } as const + return (issues ?? []) + .filter( + (i) => + i.kind === ref.kind && + (i.group ?? '') === ref.group && + (i.namespace ?? '') === ref.namespace && + i.name === ref.name, + ) + .sort((a, b) => rank[a.severity] - rank[b.severity]) +} + +export function problemsForObject(issues: CNPGWorkspaceIssue[] | undefined, ref: CNPGObjectRef): CNPGProblem[] { + return issuesForObject(issues, ref).map((issue) => ({ + id: `${issue.id}:${issue.kind}/${issue.name}`, + severity: issue.severity, + category: cnpgIssueCategory(issue), + title: issue.message || issue.reason, + detail: issue.cause || undefined, + subject: { kind: issue.kind, group: issue.group ?? '', namespace: issue.namespace ?? '', name: issue.name }, + source: 'issue', + })) +} + +// --------------------------------------------------------------------------- +// Backups and schedules +// --------------------------------------------------------------------------- + +/** + * The ScheduledBackup that created a Backup. The owner reference is only set + * when the schedule's `backupOwnerReference` is `self`; the operator labels + * every Backup it creates from a schedule regardless, so the label is read too. + */ +export function scheduledBackupOf(backup: any): string | null { + const refs = backup?.metadata?.ownerReferences + if (Array.isArray(refs)) { + const owner = refs.find((r: any) => r?.kind === 'ScheduledBackup' && isApiGroup(r?.apiVersion, CNPG_GROUP)) + if (owner?.name) return owner.name + } + const label = backup?.metadata?.labels?.[SCHEDULED_BACKUP_LABEL] + return typeof label === 'string' && label ? label : null +} + +/** Backups a ScheduledBackup created, newest first. */ +export function backupsForScheduledBackup(schedule: any, backups: any[]): any[] { + const ns = nsOf(schedule) + const name = nameOf(schedule) + return backups + .filter((b) => isCNPGKind(b, 'Backup') && nsOf(b) === ns && scheduledBackupOf(b) === name) + .sort((a, b) => backupTime(b) - backupTime(a)) +} + +export function backupTime(backup: any): number { + return parseTime(backup?.status?.startedAt) || parseTime(backup?.metadata?.creationTimestamp) +} + +/** + * The ObjectStore a plugin Backup wrote to: its own plugin parameters first, + * then the target Cluster's barman-cloud plugin. Null for non-plugin methods. + */ +export function objectStoreForBackup(backup: any, clusters: any[]): string | null { + const method = backup?.status?.method || backup?.spec?.method + if (method !== 'plugin') return null + const cfg = backup?.spec?.pluginConfiguration + const own = cfg?.parameters?.barmanObjectName + if (typeof own === 'string' && own) return own + if (cfg?.name && cfg.name !== CNPG_BARMAN_PLUGIN_NAME) return null + const cluster = targetCluster(backup, clusters) + return (cluster && getCNPGClusterBarmanPlugin(cluster)?.barmanObjectName) || null +} + +export type CNPGBackupDestination = + | { type: 'objectStore'; name: string } + | { type: 'path'; path: string } + | { type: 'volumeSnapshot' } + | { type: 'unknown' } + +export function backupDestination(backup: any, clusters: any[]): CNPGBackupDestination { + const store = objectStoreForBackup(backup, clusters) + if (store) return { type: 'objectStore', name: store } + const method = backup?.status?.method || backup?.spec?.method + if (method === 'volumeSnapshot') return { type: 'volumeSnapshot' } + const path = backup?.status?.destinationPath + if (typeof path === 'string' && path) return { type: 'path', path } + return { type: 'unknown' } +} + +// --------------------------------------------------------------------------- +// ObjectStore +// --------------------------------------------------------------------------- + +export interface CNPGObjectStoreUser { + cluster: any + /** Key of this cluster's archive inside the store's recovery windows. */ + serverName: string +} + +/** Clusters in the store's namespace whose barman-cloud plugin archives to it. */ +export function usersOfObjectStore(store: any, clusters: any[]): CNPGObjectStoreUser[] { + const ns = nsOf(store) + const name = nameOf(store) + const out: CNPGObjectStoreUser[] = [] + for (const c of clusters) { + if (!isCNPGKind(c, 'Cluster') || nsOf(c) !== ns) continue + const plugin = getCNPGClusterBarmanPlugin(c) + if (plugin?.barmanObjectName !== name) continue + out.push({ cluster: c, serverName: plugin.serverName || nameOf(c) }) + } + return out.sort((a, b) => nameOf(a.cluster).localeCompare(nameOf(b.cluster))) +} + +export interface CNPGObjectStoreEvidence { + cluster: CNPGObjectRef + serverName: string + archiving: CNPGFact + window: CNPGObjectStoreRecoveryWindow | null +} + +export interface CNPGObjectStoreHealth { + summary: CNPGFact + evidence: CNPGObjectStoreEvidence[] +} + +function archivingFact(cluster: any): CNPGFact { + const conds = cluster?.status?.conditions + const c = Array.isArray(conds) ? conds.find((x: any) => x?.type === 'ContinuousArchiving') : null + if (!c) return { text: 'WAL archiving not reported', tone: 'unknown' } + if (c.status === 'True') return { text: 'WAL archiving', tone: 'healthy', at: c.lastTransitionTime } + if (c.status === 'False') { + return { text: c.message ? `WAL archiving failing · ${c.message}` : 'WAL archiving failing', tone: 'unhealthy', at: c.lastTransitionTime } + } + return { text: 'WAL archiving unknown', tone: 'unknown' } +} + +/** + * Upload health for an ObjectStore, inferred from the clusters that use it. + * ObjectStore publishes no health of its own, so the only evidence is each + * user cluster's ContinuousArchiving condition and whether the store's + * recovery window for that cluster records a failure newer than its last + * success. + */ +export function inferredObjectStoreHealth(store: any, users: CNPGObjectStoreUser[]): CNPGObjectStoreHealth { + const windows = getCNPGObjectStoreRecoveryWindows(store) + const evidence: CNPGObjectStoreEvidence[] = users.map((u) => ({ + cluster: refOf(u.cluster, 'Cluster'), + serverName: u.serverName, + archiving: archivingFact(u.cluster), + window: windows.find((w) => w.server === u.serverName) ?? null, + })) + if (evidence.length === 0) { + return { summary: { text: 'No visible cluster uses this store', tone: 'unknown' }, evidence } + } + const failing = evidence.filter((e) => e.archiving.tone === 'unhealthy' || e.window?.failingSinceLastSuccess) + if (failing.length > 0) { + const latestFailure = failing + .map((e) => e.window?.lastFailedBackupTime ?? (e.archiving.tone === 'unhealthy' ? e.archiving.at : undefined)) + .filter((t): t is string => !!t) + .sort((a, b) => parseTime(b) - parseTime(a))[0] + return { + summary: { + text: failing.length === evidence.length ? 'Uploads failing' : `Uploads failing for ${failing.length} of ${evidence.length} clusters`, + tone: 'unhealthy', + at: latestFailure, + }, + evidence, + } + } + if (evidence.every((e) => e.archiving.tone === 'healthy')) { + return { summary: { text: 'Uploads succeeding', tone: 'healthy' }, evidence } + } + return { summary: { text: 'No failures reported', tone: 'unknown' }, evidence } +} + +// --------------------------------------------------------------------------- +// Declarative objects +// --------------------------------------------------------------------------- + +export function appliedFact(obj: any): CNPGFact { + const applied = obj?.status?.applied + if (applied === true) return { text: 'Applied', tone: 'healthy' } + if (applied === false) return { text: 'Not applied', tone: 'unhealthy' } + return { text: 'Pending · the operator has not reported a result yet', tone: 'unknown' } +} + +export function observedGenerationFact(obj: any): CNPGFact { + const observed = obj?.status?.observedGeneration + const generation = obj?.metadata?.generation + if (typeof observed !== 'number') return { text: 'Not reported', tone: 'unknown' } + if (typeof generation !== 'number') return { text: `Generation ${observed}`, tone: 'neutral' } + if (observed >= generation) return { text: `Current · generation ${generation}`, tone: 'neutral' } + return { text: `Behind · observed ${observed}, declared ${generation}`, tone: 'degraded' } +} + +/** + * A role the operator says is missing, when that role is also absent from the + * target Cluster's managed roles. Only an adjacent fact: roles may be created + * outside `spec.managed.roles`. + */ +export function missingManagedRole(obj: any, cluster: any | null): string | null { + if (!cluster) return null + const msg = obj?.status?.message + if (typeof msg !== 'string') return null + const m = msg.match(/role "([^"]+)" does not exist/) + if (!m) return null + const roles = cluster?.spec?.managed?.roles + const names = Array.isArray(roles) ? roles.map((r: any) => r?.name) : [] + return names.includes(m[1]) ? null : m[1] +} + +export { cnpgGitOpsSource as gitopsSourceOf } from './workspace' + +/** Publications and Subscriptions on the same Cluster and PostgreSQL database. */ +export function replicationForDatabase( + database: any, + publications: any[], + subscriptions: any[], +): { publications: any[]; subscriptions: any[] } { + const ns = nsOf(database) + const cluster = specCluster(database) + const dbname = database?.spec?.name + const match = (o: any, kind: string) => + isCNPGKind(o, kind) && !!cluster && nsOf(o) === ns && specCluster(o) === cluster && !!dbname && o?.spec?.dbname === dbname + return { + publications: publications.filter((p) => match(p, 'Publication')), + subscriptions: subscriptions.filter((s) => match(s, 'Subscription')), + } +} + +/** The Database object declaring the PostgreSQL database a Publication/Subscription runs in. */ +export function databaseForDeclaration(obj: any, databases: any[]): any | null { + const cluster = specCluster(obj) + const dbname = obj?.spec?.dbname + if (!cluster || !dbname) return null + return ( + databases.find( + (d) => isCNPGKind(d, 'Database') && nsOf(d) === nsOf(obj) && specCluster(d) === cluster && d?.spec?.name === dbname, + ) ?? null + ) +} + +// --------------------------------------------------------------------------- +// Image catalogs +// --------------------------------------------------------------------------- + +export interface CNPGImageCatalogUser { + cluster: any + major: number | null +} + +/** + * Clusters pinned to a catalog through `spec.imageCatalogRef`. An ImageCatalog + * is namespace-local; a ClusterImageCatalog is referenceable from any + * namespace. A ref without `kind` means ImageCatalog. + */ +export function clustersUsingCatalog(catalog: any, clusters: any[]): CNPGImageCatalogUser[] { + const kind = catalog?.kind + if (kind !== 'ImageCatalog' && kind !== 'ClusterImageCatalog') return [] + if (!isApiGroup(catalog?.apiVersion, CNPG_GROUP)) return [] + const name = nameOf(catalog) + return clusters + .filter((c) => { + if (!isCNPGKind(c, 'Cluster')) return false + const ref = c?.spec?.imageCatalogRef + if (!ref?.name || ref.name !== name) return false + if ((ref.kind || 'ImageCatalog') !== kind) return false + return kind === 'ClusterImageCatalog' || nsOf(c) === nsOf(catalog) + }) + .map((c) => { + const major = c?.spec?.imageCatalogRef?.major + return { cluster: c, major: typeof major === 'number' ? major : null } + }) + .sort((a, b) => nsOf(a.cluster).localeCompare(nsOf(b.cluster)) || nameOf(a.cluster).localeCompare(nameOf(b.cluster))) +} diff --git a/packages/k8s-ui/src/components/cnpg/workspace.test.ts b/packages/k8s-ui/src/components/cnpg/workspace.test.ts index 5004ec3bb..75d40b444 100644 --- a/packages/k8s-ui/src/components/cnpg/workspace.test.ts +++ b/packages/k8s-ui/src/components/cnpg/workspace.test.ts @@ -176,4 +176,58 @@ describe('buildCNPGFleet', () => { expect(d.failed).toBe(1) expect(d.summary.tone).toBe('degraded') }) + + it('treats declared managed roles without status as pending, not reconciled', () => { + const c = cluster('pg-a', 'db', { spec: { managed: { roles: [{ name: 'app' }, { name: 'audit' }] } } }) + const d = buildCNPGFleet(resp({ clusters: [c] })).rows[0].declarations + expect(d.pending).toBe(2) + expect(d.summary.text).toBe('2 of 2 pending') + }) + + it('does not claim "no schedule" when ScheduledBackups are unreadable', () => { + const fleet = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { scheduledBackups: { state: 'denied' } } })) + expect(fleet.rows[0].protection.summary.text).not.toBe('No backup destination or schedule') + }) + + it('says recovery is only declared until the restored cluster has a ready instance', () => { + const src = cluster('pg-a', 'db') + const restoring = cluster('pg-a-restore', 'db', { + spec: { bootstrap: { recovery: { backup: { name: 'nightly-1' } } } }, + status: { readyInstances: 0, currentPrimary: undefined }, + }) + const backups = [{ apiVersion: G, kind: 'Backup', metadata: { name: 'nightly-1', namespace: 'db' }, spec: { cluster: { name: 'pg-a' } }, status: { phase: 'completed' } }] + const a = buildCNPGFleet(resp({ clusters: [src, restoring], backups })).rows.find((r) => r.name === 'pg-a')! + expect(a.protection.restoreValidation.text).toBe('Recovery declared in pg-a-restore') + expect(a.protection.restoreValidation.tone).toBe('unknown') + }) + + it('does not attribute a restore by backup-name prefix alone', () => { + const src = cluster('pg', 'db') + const other = cluster('pg-orders-restore', 'db', { spec: { bootstrap: { recovery: { backup: { name: 'pg-orders-backup' } } } } }) + const backups = [{ apiVersion: G, kind: 'Backup', metadata: { name: 'pg-orders-backup', namespace: 'db' }, spec: { cluster: { name: 'pg-orders' } }, status: { phase: 'completed' } }] + const row = buildCNPGFleet(resp({ clusters: [src, other], backups })).rows.find((r) => r.name === 'pg')! + expect(row.protection.restoreValidation.text).toBe('None recorded') + }) + + it('marks Poolers unknown when they are not readable', () => { + const fleet = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { poolers: { state: 'denied' } } })) + expect(fleet.rows[0].poolersKnown).toBe(false) + }) + + it('attributes instance Pod issues to their cluster', () => { + const fleet = buildCNPGFleet( + resp({ clusters: [cluster('pg-a', 'db')], pods: [pod('pg-a-2', 'db', 'pg-a', 'replica', false)] }, { + issues: [{ id: 'p1', severity: 'critical', kind: 'Pod', namespace: 'db', name: 'pg-a-2', reason: 'CrashLoopBackOff', message: 'Back-off restarting failed container' }], + }), + ) + expect(fleet.rows[0].attention).toBe(true) + expect(fleet.rows[0].categories.has('availability')).toBe(true) + }) + + it('reads partial coverage by allowed namespaces and treats unnamed partial coverage as unknown', () => { + const named = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'partial', allowedNamespaces: ['db'] } } })) + expect(named.rows[0].protection.lastSuccessfulBackup.text).toBe('None observed') + const unnamed = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'partial' } } })) + expect(unnamed.rows[0].protection.lastSuccessfulBackup.text).toBe('No access to Backups') + }) }) diff --git a/packages/k8s-ui/src/components/cnpg/workspace.ts b/packages/k8s-ui/src/components/cnpg/workspace.ts index 6a332dee9..7276063a8 100644 --- a/packages/k8s-ui/src/components/cnpg/workspace.ts +++ b/packages/k8s-ui/src/components/cnpg/workspace.ts @@ -32,7 +32,10 @@ export type CNPGCoverageState = 'full' | 'partial' | 'denied' | 'notInstalled' | export interface CNPGKindCoverage { state: CNPGCoverageState + /** Denied namespaces, named only when the caller supplied the candidate list. */ deniedNamespaces?: string[] + /** For partial coverage: the namespaces that were read. */ + allowedNamespaces?: string[] } export interface CNPGWorkspaceIssue { @@ -158,12 +161,14 @@ export interface CNPGFleetRow { protection: CNPGProtectionFacts & { summary: CNPGFact } declarations: { summary: CNPGFact; total: number; failed: number; pending: number } poolers: string[] + /** False when Poolers are not readable in this cluster's namespace, so an empty list means unknown. */ + poolersKnown: boolean problems: CNPGProblem[] /** Any issue at warning or worse. Posture findings alone do not need attention. */ attention: boolean categories: Set /** GitOps owner recorded on the Cluster, when it carries the standard labels. */ - gitops: { tool: 'argocd' | 'flux'; name: string; namespace?: string } | null + gitops: CNPGGitOpsSource | null } export interface CNPGFleet { @@ -181,7 +186,7 @@ const PROTECTION_ISSUE_REASONS = new Set([ 'CNPGScheduledBackupMissed', ]) -function issueCategory(issue: CNPGWorkspaceIssue): CNPGProblemCategory { +export function cnpgIssueCategory(issue: Pick): CNPGProblemCategory { if (PROTECTION_ISSUE_REASONS.has(issue.reason)) return 'protection' switch (issue.kind) { case 'Backup': @@ -215,7 +220,12 @@ function coverageOf(resp: CNPGWorkspaceResponse, k: CNPGWorkspaceKey): CNPGKindC /** Coverage is usable in a namespace when that namespace's objects were read. */ export function coverageReadable(cov: CNPGKindCoverage, namespace?: string): boolean { if (cov.state === 'full') return true - if (cov.state === 'partial') return !namespace || !(cov.deniedNamespaces ?? []).includes(namespace) + if (cov.state === 'partial') { + if (!namespace) return false + if (cov.allowedNamespaces) return cov.allowedNamespaces.includes(namespace) + if (cov.deniedNamespaces) return !cov.deniedNamespaces.includes(namespace) + return false + } return false } @@ -256,9 +266,12 @@ function podReady(pod: any): boolean | null { return ready.status === 'True' } -function clusterGitOps(cluster: any): CNPGFleetRow['gitops'] { - const labels = cluster?.metadata?.labels ?? {} - const annotations = cluster?.metadata?.annotations ?? {} +export type CNPGGitOpsSource = { tool: 'argocd' | 'flux'; name: string; namespace?: string } + +/** The GitOps owner recorded on an object's standard Argo CD / Flux labels. */ +export function cnpgGitOpsSource(obj: any): CNPGGitOpsSource | null { + const labels = obj?.metadata?.labels ?? {} + const annotations = obj?.metadata?.annotations ?? {} const argo = labels['argocd.argoproj.io/instance'] if (argo) return { tool: 'argocd', name: argo } const tracking = annotations['argocd.argoproj.io/tracking-id'] @@ -387,36 +400,50 @@ function walFact(cluster: any): CNPGFact { function restoreValidationFact( cluster: any, allClusters: any[], + backups: any[], ): CNPGProtectionFacts['restoreValidation'] { const plugin = getCNPGClusterBarmanPlugin(cluster) const server = plugin?.serverName || cluster.metadata?.name const store = plugin?.barmanObjectName const ns = cluster.metadata?.namespace + const name = cluster.metadata?.name const restored = allClusters.find((c) => { if (c === cluster || c.metadata?.namespace !== ns) return false const recovery = c.spec?.bootstrap?.recovery if (!recovery) return false const sourceName = recovery.source - const ext = (c.spec?.externalClusters ?? []).find((e: any) => e?.name === sourceName) - const params = ext?.plugin?.parameters - if (store && params?.barmanObjectName === store && (params?.serverName || sourceName) === server) return true + if (sourceName) { + const ext = (c.spec?.externalClusters ?? []).find((e: any) => e?.name === sourceName) + const params = ext?.plugin?.parameters + if (store && params?.barmanObjectName === store && (params?.serverName || sourceName) === server) return true + } const backupName = recovery.backup?.name - return !!backupName && typeof backupName === 'string' && backupName.startsWith(`${cluster.metadata?.name}-`) + if (!backupName) return false + const backup = backups.find((b) => b.metadata?.namespace === ns && b.metadata?.name === backupName) + return specClusterName(backup) === name }) - if (restored) { + if (!restored) return { text: 'None recorded', tone: 'unknown', source: 'Kubernetes does not record restore tests' } + const rname = restored.metadata?.name + const ready = typeof restored.status?.readyInstances === 'number' && restored.status.readyInstances > 0 + if (!ready) { return { - text: `Restored into ${restored.metadata?.name}`, - tone: 'neutral', - source: `Cluster ${restored.metadata?.name} bootstrapped from this cluster's backups · created ${restored.metadata?.creationTimestamp ?? 'unknown'}`, - restoredInto: { namespace: restored.metadata?.namespace, name: restored.metadata?.name }, + text: `Recovery declared in ${rname}`, + tone: 'unknown', + source: `Cluster ${rname} bootstraps from this cluster's backups but has no ready instance yet`, + restoredInto: { namespace: restored.metadata?.namespace, name: rname }, } } - return { text: 'None recorded', tone: 'unknown', source: 'Kubernetes does not record restore tests' } + return { + text: `Restored into ${rname}`, + tone: 'neutral', + source: `Cluster ${rname} bootstrapped from this cluster's backups and has ready instances · created ${restored.metadata?.creationTimestamp ?? 'unknown'}. This proves one recovery, not that today's backups restore.`, + restoredInto: { namespace: restored.metadata?.namespace, name: rname }, + } } function protectionSummary(p: CNPGProtectionFacts): CNPGFact { if (p.walArchiving.tone === 'unhealthy') return { text: 'WAL archiving failing', tone: 'unhealthy' } - if (p.destination.method === 'none' && p.schedule.names.length === 0) { + if (p.destination.method === 'none' && p.schedule.names.length === 0 && p.schedule.tone !== 'unknown') { return { text: 'No backup destination or schedule', tone: 'neutral' } } if (p.schedule.tone === 'degraded') return { text: p.schedule.text, tone: 'degraded' } @@ -469,9 +496,9 @@ function problemsFor( const owner = children.get(`${issue.kind}/${ns}/${issue.name}`) if (!isSelf && owner !== name) continue out.push({ - id: issue.id, + id: `${issue.id}:${issue.kind}/${issue.name}`, severity: issue.severity, - category: issueCategory(issue), + category: cnpgIssueCategory(issue), title: issue.message || issue.reason, detail: issue.cause || undefined, subject: { kind: issue.kind, group: issue.group ?? '', namespace: ns, name: issue.name }, @@ -509,6 +536,10 @@ function childIndex(resp: CNPGWorkspaceResponse): Map { add('Database', resp.objects.databases) add('Publication', resp.objects.publications) add('Subscription', resp.objects.subscriptions) + for (const p of resp.objects.pods ?? []) { + const c = p?.metadata?.labels?.['cnpg.io/cluster'] + if (c) idx.set(`Pod/${p.metadata?.namespace}/${p.metadata?.name}`, c) + } return idx } @@ -536,11 +567,16 @@ function declarationsFor(cluster: any, resp: CNPGWorkspaceResponse): CNPGFleetRo else if (o.status?.applied !== true) pending++ } } - const roles = cluster?.status?.managedRolesStatus - const roleErrors = roles?.cannotReconcile ? Object.keys(roles.cannotReconcile).length : 0 - const declaredRoles = Array.isArray(cluster?.spec?.managed?.roles) ? cluster.spec.managed.roles.length : 0 - failed += roleErrors - total += declaredRoles + const roleStatus = cluster?.status?.managedRolesStatus + const reconciledRoles = new Set(roleStatus?.byStatus?.reconciled ?? []) + const failedRoles = new Set(Object.keys(roleStatus?.cannotReconcile ?? {})) + const declaredRoles: any[] = Array.isArray(cluster?.spec?.managed?.roles) ? cluster.spec.managed.roles : [] + for (const r of declaredRoles) { + if (!r?.name) continue + total++ + if (failedRoles.has(r.name)) failed++ + else if (!reconciledRoles.has(r.name)) pending++ + } let summary: CNPGFact if (total === 0) { summary = unreadable ? { text: 'No access to some declarations', tone: 'unknown' } : { text: 'None declared', tone: 'neutral' } @@ -594,7 +630,7 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { source: `ObjectStore ${window.store} status`, } : { text: 'Not reported', tone: 'unknown' }, - restoreValidation: restoreValidationFact(cluster, clusters), + restoreValidation: restoreValidationFact(cluster, clusters, resp.objects.backups ?? []), } const problems = problemsFor(cluster, resp.issues ?? [], resp.audit ?? [], children) const categories = new Set( @@ -620,10 +656,11 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { poolers: poolers .filter((p) => p.metadata?.namespace === ns && specClusterName(p) === name) .map((p) => p.metadata?.name), + poolersKnown: coverageReadable(coverageOf(resp, 'poolers'), ns), problems, attention: problems.some((p) => p.severity !== 'posture'), categories, - gitops: clusterGitOps(cluster), + gitops: cnpgGitOpsSource(cluster), } }) diff --git a/web/src/api/cnpg.ts b/web/src/api/cnpg.ts index d73971493..5ccace5e3 100644 --- a/web/src/api/cnpg.ts +++ b/web/src/api/cnpg.ts @@ -19,3 +19,51 @@ export function useCNPGWorkspace(namespaces: string[], options?: { enabled?: boo placeholderData: (prev) => prev, }) } + +export interface CNPGOperatorCoverage { + state: 'full' | 'partial' | 'denied' | 'syncing' | 'error' + deniedNamespaces?: string[] +} + +export interface CNPGOperatorComponent { + role: 'operator' | 'plugin' + pluginName?: string + namespace: string + deployment: string + image?: string + version?: string + readyReplicas: number | null + replicas: number | null +} + +export interface CNPGOperatorConfig { + kind: 'ConfigMap' | 'Secret' + namespace: string + name: string + purpose: 'operator' | 'monitoring' + exists?: boolean | null + readable?: boolean + reason?: string + data?: Record +} + +export interface CNPGOperatorResponse { + coverage: { deployments: CNPGOperatorCoverage; services: CNPGOperatorCoverage } + components: CNPGOperatorComponent[] + config: CNPGOperatorConfig[] +} + +// /api/cnpg/operator +// +// Operator and plugin workloads plus where the operator's configuration lives. +// Deliberately not filtered by the namespace view filter: the operator runs in +// its own namespace, which users rarely have selected. +export function useCNPGOperator(options?: { enabled?: boolean }) { + return useQuery({ + queryKey: ['cnpg', 'operator'], + queryFn: ({ signal }) => fetchJSON('/cnpg/operator', signal), + enabled: options?.enabled ?? true, + staleTime: 30_000, + refetchInterval: 60_000, + }) +} diff --git a/web/src/components/cnpg/CNPGDeclarations.tsx b/web/src/components/cnpg/CNPGDeclarations.tsx new file mode 100644 index 000000000..254db48bf --- /dev/null +++ b/web/src/components/cnpg/CNPGDeclarations.tsx @@ -0,0 +1,303 @@ +import { useMemo, type ReactNode } from 'react' +import { clsx } from 'clsx' +import { AlertTriangle } from 'lucide-react' +import { Badge, isApiGroup, toneTextClass, type CNPGFleetRow } from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { + CNPGWorkspaceHeader, + CoverageNotice, + FilterChips, + ScreenBody, + Segments, + Sub, + clusterResource, + cnpgResource, + namespaceChip, + type CNPGScreenProps, +} from './shared' +import { sameResource } from './routes' + +type State = 'applied' | 'failed' | 'pending' + +interface DeclItem { + key: string + kind: 'Database' | 'Publication' | 'Subscription' | 'Managed role' + pgName: string + indent: boolean + state: State + meta?: string + error?: string + source?: string + resource: SelectedResource + isField: boolean +} + +interface DeclGroup { + namespace: string + cluster: string + row: CNPGFleetRow | null + items: DeclItem[] +} + +function stateOf(obj: any): State { + const applied = obj?.status?.applied + if (applied === true) return 'applied' + if (applied === false) return 'failed' + return 'pending' +} + +function gitopsSource(obj: any): string | undefined { + const labels = obj?.metadata?.labels ?? {} + if (labels['argocd.argoproj.io/instance']) return `Argo CD ${labels['argocd.argoproj.io/instance']}` + const flux = labels['kustomize.toolkit.fluxcd.io/name'] || labels['helm.toolkit.fluxcd.io/name'] + if (flux) return `Flux ${flux}` + return undefined +} + +const STATE_BADGE: Record = { + applied: { severity: 'success', text: 'Applied' }, + failed: { severity: 'warning', text: 'Not applied' }, + pending: { severity: 'neutral', text: 'Pending' }, +} + +function roleState(cluster: any, role: string): { state: State; error?: string } { + const status = cluster?.status?.managedRolesStatus + const errs = status?.cannotReconcile?.[role] + if (Array.isArray(errs) && errs.length > 0) return { state: 'failed', error: errs.join('; ') } + const by = status?.byStatus ?? {} + if ((by.reconciled ?? []).includes(role)) return { state: 'applied' } + return { state: 'pending' } +} + +export function CNPGDeclarations({ data, fleet, namespaces, searchParams, onSetParams, onInspect, inspected, onClearNamespaces }: CNPGScreenProps) { + const clusterFilter = searchParams.get('cluster') + const onlyFailed = searchParams.get('show') === 'failed' + + const groups = useMemo(() => { + const byCluster = new Map() + const group = (ns: string, cluster: string) => { + const k = `${ns}/${cluster}` + let g = byCluster.get(k) + if (!g) { + g = { namespace: ns, cluster, row: fleet.rows.find((r) => r.namespace === ns && r.name === cluster) ?? null, items: [] } + byCluster.set(k, g) + } + return g + } + const valid = (o: any) => isApiGroup(o?.apiVersion, 'postgresql.cnpg.io') + const dbs = (data.objects.databases ?? []).filter(valid) + const pubs = (data.objects.publications ?? []).filter(valid) + const subs = (data.objects.subscriptions ?? []).filter(valid) + const placed = new Set() + for (const d of dbs) { + const ns = d.metadata?.namespace ?? '' + const cluster = d.spec?.cluster?.name ?? '(no cluster)' + const g = group(ns, cluster) + const st = stateOf(d) + g.items.push({ + key: `db/${ns}/${d.metadata?.name}`, + kind: 'Database', + pgName: d.spec?.name ?? d.metadata?.name, + indent: false, + state: st, + meta: d.spec?.owner ? `owner ${d.spec.owner}` : undefined, + error: st === 'failed' ? d.status?.message : undefined, + source: gitopsSource(d), + resource: cnpgResource('databases', ns, d.metadata?.name), + isField: false, + }) + for (const [list, kind] of [[pubs, 'Publication'], [subs, 'Subscription']] as const) { + for (const p of list) { + if (p.metadata?.namespace !== ns || p.spec?.cluster?.name !== d.spec?.cluster?.name || p.spec?.dbname !== d.spec?.name) continue + placed.add(p) + const pst = stateOf(p) + g.items.push({ + key: `${kind}/${ns}/${p.metadata?.name}`, + kind, + pgName: p.spec?.name ?? p.metadata?.name, + indent: true, + state: pst, + meta: + kind === 'Publication' + ? p.spec?.target?.allTables ? 'all tables' : 'selected objects' + : `from ${p.spec?.publicationName ?? '?'} on ${p.spec?.externalClusterName ?? '?'}`, + error: pst === 'failed' ? p.status?.message : undefined, + source: gitopsSource(p), + resource: cnpgResource(kind === 'Publication' ? 'publications' : 'subscriptions', ns, p.metadata?.name), + isField: false, + }) + } + } + } + for (const [list, kind] of [[pubs, 'Publication'], [subs, 'Subscription']] as const) { + for (const p of list) { + if (placed.has(p)) continue + const ns = p.metadata?.namespace ?? '' + const g = group(ns, p.spec?.cluster?.name ?? '(no cluster)') + const pst = stateOf(p) + g.items.push({ + key: `${kind}/${ns}/${p.metadata?.name}`, + kind, + pgName: p.spec?.name ?? p.metadata?.name, + indent: false, + state: pst, + meta: p.spec?.dbname ? `database ${p.spec.dbname}` : undefined, + error: pst === 'failed' ? p.status?.message : undefined, + source: gitopsSource(p), + resource: cnpgResource(kind === 'Publication' ? 'publications' : 'subscriptions', ns, p.metadata?.name), + isField: false, + }) + } + } + for (const row of fleet.rows) { + const roles: any[] = Array.isArray(row.cluster?.spec?.managed?.roles) ? row.cluster.spec.managed.roles : [] + if (roles.length === 0) continue + const g = group(row.namespace, row.name) + for (const r of roles) { + if (!r?.name) continue + const rs = roleState(row.cluster, r.name) + g.items.push({ + key: `role/${row.namespace}/${row.name}/${r.name}`, + kind: 'Managed role', + pgName: r.name, + indent: false, + state: rs.state, + meta: r.ensure === 'absent' ? 'ensure absent' : 'spec.managed.roles', + error: rs.error, + source: `Cluster ${row.name} spec`, + resource: clusterResource(row.namespace, row.name), + isField: true, + }) + } + } + return [...byCluster.values()] + .filter((g) => !clusterFilter || `${g.namespace}/${g.cluster}` === clusterFilter) + .map((g) => ({ ...g, items: onlyFailed ? g.items.filter((i) => i.state === 'failed') : g.items })) + .filter((g) => g.items.length > 0) + .sort((a, b) => { + const fa = a.items.some((i) => i.state === 'failed') ? 0 : 1 + const fb = b.items.some((i) => i.state === 'failed') ? 0 : 1 + return fa - fb || a.namespace.localeCompare(b.namespace) || a.cluster.localeCompare(b.cluster) + }) + }, [data.objects.databases, data.objects.publications, data.objects.subscriptions, fleet.rows, clusterFilter, onlyFailed]) + + const failedTotal = useMemo(() => { + let n = 0 + for (const k of ['databases', 'publications', 'subscriptions'] as const) { + n += (data.objects[k] ?? []).filter((o) => o?.status?.applied === false).length + } + for (const r of fleet.rows) n += Object.keys(r.cluster?.status?.managedRolesStatus?.cannotReconcile ?? {}).length + return n + }, [data.objects, fleet.rows]) + + const chips = [ + ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), + ...namespaceChip(namespaces, onClearNamespaces), + ] + + return ( +
+ + + +
+ onSetParams({ show: id === 'failed' ? 'failed' : null })} + options={[ + { id: 'all', label: 'All declarations' }, + { id: 'failed', label: 'Not reconciled', count: failedTotal }, + ]} + /> +
+ + + {groups.length === 0 ? ( +
+ {onlyFailed ? 'Every declaration in this scope is reconciled.' : 'No declarations in this scope.'} +
+ ) : ( + groups.map((g) => { + const failed = g.items.filter((i) => i.state === 'failed').length + return ( +
+
+ {g.row ? ( + + ) : ( + + {g.cluster} + + )} + {g.namespace} + 0 ? toneTextClass('degraded') : 'text-theme-text-tertiary')}> + {g.items.length} {g.items.length === 1 ? 'declaration' : 'declarations'} + {failed > 0 ? ` · ${failed} not reconciled` : ''} + {!g.row ? ' · target cluster not visible' : ''} + +
+
+ {g.items.map((i) => ( + onInspect(i.resource)} + /> + ))} +
+
+ ) + }) + )} +
+
+ ) +} + +function DeclarationRow({ item, active, onInspect }: { item: DeclItem; active: boolean; onInspect: () => void }) { + const badge = STATE_BADGE[item.state] + let detail: ReactNode = null + if (item.error) { + detail = ( +
+ + Controller error: {item.error} +
+ ) + } + return ( +
{ if (e.key === 'Enter') onInspect() }} + className={clsx( + 'grid cursor-pointer grid-cols-[minmax(0,1.6fr)_minmax(0,0.8fr)_minmax(0,1fr)] gap-x-4 px-4 py-2.5 text-sm transition-colors hover:bg-theme-hover/50', + active && 'selection', + )} + > +
+
+ {item.kind} + {item.pgName} + {item.meta && {item.meta}} +
+ {detail} +
+
+ {badge.text} + {item.isField && field of the Cluster} +
+
+ {item.source ?? Applied directly (no GitOps owner label)} +
+
+ ) +} diff --git a/web/src/components/cnpg/CNPGOperator.tsx b/web/src/components/cnpg/CNPGOperator.tsx new file mode 100644 index 000000000..2bf3e59e4 --- /dev/null +++ b/web/src/components/cnpg/CNPGOperator.tsx @@ -0,0 +1,226 @@ +import { useMemo } from 'react' +import { Badge, getCNPGImageCatalogEntries, isApiGroup, PaneLoader } from '@skyhook-io/k8s-ui' +import { useCNPGOperator, type CNPGOperatorComponent, type CNPGOperatorConfig } from '../../api/cnpg' +import { Notice } from '../capacity/shared' +import { + CNPGWorkspaceHeader, + Mono, + ScreenBody, + SectionTable, + Sub, + cnpgResource, + type CNPGScreenProps, +} from './shared' + +interface CatalogRow { + key: string + kind: 'ImageCatalog' | 'ClusterImageCatalog' + namespace: string + name: string + images: { major: number; image: string }[] + users: { namespace: string; name: string; major?: number }[] +} + +function readiness(c: CNPGOperatorComponent) { + if (c.readyReplicas === null || c.replicas === null) return Unknown + if (c.replicas === 0) return Scaled to 0 + const ok = c.readyReplicas >= c.replicas + return {c.readyReplicas}/{c.replicas} ready +} + +export function CNPGOperator({ data, fleet, onInspect, inspected }: CNPGScreenProps) { + const operator = useCNPGOperator() + + const catalogs = useMemo(() => { + const out: CatalogRow[] = [] + const clusters = fleet.rows + for (const [key, kind] of [['imageCatalogs', 'ImageCatalog'], ['clusterImageCatalogs', 'ClusterImageCatalog']] as const) { + for (const cat of data.objects[key] ?? []) { + if (!isApiGroup(cat.apiVersion, 'postgresql.cnpg.io')) continue + const ns = cat.metadata?.namespace ?? '' + const name = cat.metadata?.name ?? '' + const users = clusters + .filter((r) => { + const ref = r.cluster?.spec?.imageCatalogRef + if (!ref || ref.name !== name) return false + const refKind = ref.kind || 'ImageCatalog' + if (refKind !== kind) return false + return kind === 'ClusterImageCatalog' || r.namespace === ns + }) + .map((r) => ({ namespace: r.namespace, name: r.name, major: r.cluster?.spec?.imageCatalogRef?.major })) + out.push({ key: `${kind}/${ns}/${name}`, kind, namespace: ns, name, images: getCNPGImageCatalogEntries(cat), users }) + } + } + return out + }, [data.objects, fleet.rows]) + + const direct = fleet.rows.filter((r) => !r.cluster?.spec?.imageCatalogRef) + const op = operator.data + const coverageGaps = op + ? (['deployments', 'services'] as const).filter((k) => op.coverage[k]?.state !== 'full') + : [] + + return ( +
+ + + {operator.isLoading && !op ? ( + + ) : !op ? ( + Operator details could not be loaded{operator.error instanceof Error ? `: ${operator.error.message}` : '.'} + ) : ( + <> + {coverageGaps.length > 0 && ( + + Some workloads are not readable ({coverageGaps.map((k) => `${k}: ${op.coverage[k].state}`).join(', ')}), so an operator or plugin running in those namespaces may be missing below. + + )} + ( + <> +
{c.role === 'operator' ? 'CloudNativePG operator' : c.pluginName ?? 'Plugin'}
+ {c.role === 'operator' ? 'controller manager' : 'CNPG-I plugin'} + + ), + }, + { + header: 'Version', + width: '14%', + cell: (c) => (c.version ? {c.version} : Unknown), + }, + { header: 'Ready', width: '14%', cell: readiness }, + { + header: 'Workload', + width: '42%', + cell: (c) => + c.deployment ? ( + <>Deployment {c.deployment}{c.namespace} + ) : ( + <> + No Deployment matches its Service + {c.namespace} + + ), + }, + ]} + rows={op.components} + rowKey={(c) => `${c.namespace}/${c.deployment}`} + rowResource={(c) => (c.deployment ? { kind: 'deployments', group: 'apps', namespace: c.namespace, name: c.deployment } : null)} + onInspect={onInspect} + inspected={inspected} + empty="No operator or plugin Deployments found in the namespaces you can read." + /> + + )} + + <>{c.name}{c.kind} }, + { header: 'Scope', width: '14%', cell: (c) => (c.kind === 'ClusterImageCatalog' ? 'Cluster-wide' : `Namespace ${c.namespace}`) }, + { + header: 'Images', + width: '40%', + cell: (c) => + c.images.length === 0 ? ( + None + ) : ( +
+ {c.images.map((i) => ( +
+ {i.major} + {i.image} +
+ ))} +
+ ), + }, + { + header: 'Used by', + width: '24%', + cell: (c) => + c.users.length === 0 ? ( + No visible cluster + ) : ( + c.users.map((u) => `${u.name}${u.major !== undefined ? ` (${u.major})` : ''}`).join(', ') + ), + }, + ]} + rows={catalogs} + rowKey={(c) => c.key} + rowResource={(c) => cnpgResource(c.kind === 'ImageCatalog' ? 'imagecatalogs' : 'clusterimagecatalogs', c.namespace, c.name)} + onInspect={onInspect} + inspected={inspected} + empty="No image catalogs in this scope." + footer={ + <> + Used-by lists only clusters you can see; the catalog detail asks the server for every user. + {direct.length > 0 && <> Not using a catalog (direct imageName): {direct.map((r) => r.name).join(', ')}.} + + } + /> + + {op && op.config.length > 0 && ( +
+

Operator configuration

+
+ {op.config.map((c) => ( + + ))} +
+
+ )} +
+
+ ) +} + +function ConfigBlock({ config, onInspect }: { config: CNPGOperatorConfig; onInspect: CNPGScreenProps['onInspect'] }) { + const open = () => onInspect({ kind: config.kind === 'ConfigMap' ? 'configmaps' : 'secrets', group: '', namespace: config.namespace, name: config.name }) + const title = ( +
+ {config.kind} + + + {config.namespace} · {config.purpose === 'monitoring' ? 'monitoring queries' : 'operator settings'} + +
+ ) + let body + if (config.kind === 'Secret') { + body =
Referenced by the operator. Secret contents are not shown here.
+ } else if (config.exists === false) { + body =
Referenced but does not exist; the operator runs with its defaults.
+ } else if (!config.readable) { + body =
{config.reason ?? 'Not readable with your access.'}
+ } else { + const entries = Object.entries(config.data ?? {}) + body = + entries.length === 0 ? ( +
No keys set.
+ ) : ( +
+ {entries.map(([k, v]) => ( +
+
{k}
+
{v.length > 400 ? `${v.slice(0, 400)}…` : v}
+
+ ))} +
+ ) + } + return ( +
+ {title} + {body} +
+ ) +} diff --git a/web/src/components/cnpg/CNPGOverview.tsx b/web/src/components/cnpg/CNPGOverview.tsx index d1160adde..977945b62 100644 --- a/web/src/components/cnpg/CNPGOverview.tsx +++ b/web/src/components/cnpg/CNPGOverview.tsx @@ -1,76 +1,25 @@ import { useMemo } from 'react' import { useNavigate } from 'react-router-dom' -import type { UseQueryResult } from '@tanstack/react-query' import { clsx } from 'clsx' -import { ArrowRight, Database, Search, X } from 'lucide-react' +import { ArrowRight, Database, Search } from 'lucide-react' import { - CNPG_KIND_BY_KEY, CNPG_PROBLEM_CATEGORIES, FactValue, - PaneLoader, StatusDot, Tooltip, toneTextClass, - type CNPGFleet, type CNPGFleetRow, type CNPGProblemCategory, - type CNPGWorkspaceResponse, } from '@skyhook-io/k8s-ui' import type { SelectedResource } from '../../types' import { useConnection } from '../../context/ConnectionContext' -import { EmptyState, Notice, ROW_HOVER, TABLE_HEAD, TABLE_WRAP, TBODY, TD, TH } from '../capacity/shared' +import { EmptyState, ROW_HOVER, TABLE_HEAD, TABLE_WRAP, TBODY, TD, TH } from '../capacity/shared' +import { CNPGWorkspaceHeader, CoverageNotice, FilterChips, type CNPGScreenProps } from './shared' import { cnpgClusterFullPath } from './paths' import { sameResource } from './routes' -const COVERAGE_LABEL: Record = { - denied: 'no access', - partial: 'no access in some namespaces', - syncing: 'still loading', - error: 'could not be read', -} - type Filter = 'attention' | 'all' -export function CNPGWorkspaceHeader({ - title, - subtitle, - actions, -}: { - title: string - subtitle?: React.ReactNode - actions?: React.ReactNode -}) { - return ( -
-
- - CloudNativePG -
-
-
-

{title}

- {subtitle &&
{subtitle}
} -
- {actions} -
-
- ) -} - -export function CoverageNotice({ fleet, data }: { fleet: CNPGFleet; data: CNPGWorkspaceResponse }) { - if (fleet.incompleteKinds.length === 0) return null - const parts = fleet.incompleteKinds.map((k) => { - const cov = data.coverage[k] - const label = COVERAGE_LABEL[cov?.state ?? ''] ?? cov?.state - return `${CNPG_KIND_BY_KEY[k].kind} (${label})` - }) - return ( - - Some CloudNativePG data is not readable: {parts.join(', ')}. Facts built on it read “No access” or “unknown” rather than none, and counts are lower bounds. - - ) -} - function InstancePills({ row }: { row: CNPGFleetRow }) { if (row.pods.length === 0) return null return ( @@ -106,7 +55,7 @@ function AttentionCell({ row }: { row: CNPGFleetRow }) { } export function CNPGOverview({ - query, + data, fleet, namespaces, searchParams, @@ -114,26 +63,15 @@ export function CNPGOverview({ onInspect, inspected, onClearNamespaces, -}: { - query: UseQueryResult - fleet: CNPGFleet | null - namespaces: string[] - searchParams: URLSearchParams - onSetParams: (update: Record) => void - onInspect: (resource: SelectedResource) => void - inspected: SelectedResource | null - onClearNamespaces: () => void -}) { +}: CNPGScreenProps) { const navigate = useNavigate() const { connection } = useConnection() - const data = query.data const q = searchParams.get('q') ?? '' const cat = (searchParams.get('cat') as CNPGProblemCategory | null) ?? null const rawFilter = searchParams.get('filter') as Filter | null - const filter: Filter = rawFilter ?? (fleet && fleet.attentionCount > 0 ? 'attention' : 'all') + const filter: Filter = rawFilter ?? (fleet.attentionCount > 0 ? 'attention' : 'all') const rows = useMemo(() => { - if (!fleet) return [] let list = fleet.rows if (filter === 'attention') list = list.filter((r) => r.attention) if (cat) list = list.filter((r) => r.categories.has(cat)) @@ -144,45 +82,31 @@ export function CNPGOverview({ return list }, [fleet, filter, cat, q]) - if (!data && query.isLoading) return - if (!data) { - return ( - - ) - } - if (!data.installed || !fleet) { - return ( - - ) - } - const clustersCov = data.coverage.clusters const total = fleet.rows.length const context = connection.context || data.context if (total === 0) { - const denied = clustersCov?.state === 'denied' + const state = clustersCov?.state ?? 'notInstalled' + const empty = + state === 'denied' + ? { title: 'No access to PostgreSQL clusters', detail: 'Your identity cannot list CloudNativePG Clusters. Other CloudNativePG kinds may still be browsable under Resource kinds.' } + : state === 'syncing' + ? { title: 'Loading PostgreSQL clusters', detail: 'Radar is still syncing CloudNativePG Clusters from the API server.' } + : state === 'error' + ? { title: 'PostgreSQL clusters could not be read', detail: 'Reading CloudNativePG Clusters failed; see the Radar server log.' } + : state === 'partial' + ? { title: 'No visible PostgreSQL clusters', detail: 'None in the namespaces you can read. Clusters in namespaces you cannot list are not shown.' } + : namespaces.length > 0 + ? { title: `No PostgreSQL clusters in ${context}`, detail: `None in namespace ${namespaces.join(', ')}. Clear the namespace filter to see the whole cluster.` } + : { title: `No PostgreSQL clusters in ${context}`, detail: 'The CloudNativePG CRDs are installed. Clusters, backups and declarations appear here once they exist.' } return (
0 - ? `None in namespace ${namespaces.join(', ')}. Clear the namespace filter to see the whole cluster.` - : 'The CloudNativePG CRDs are installed. Clusters, backups and declarations appear here once they exist.' - } + title={empty.title} + detail={empty.detail} action={ namespaces.length > 0 ? (
- {chips.length > 0 && ( -
- {chips.map((c) => ( - - {c.label} - - - ))} -
- )} +
diff --git a/web/src/components/cnpg/CNPGPooling.tsx b/web/src/components/cnpg/CNPGPooling.tsx new file mode 100644 index 000000000..fb18f07f0 --- /dev/null +++ b/web/src/components/cnpg/CNPGPooling.tsx @@ -0,0 +1,116 @@ +import { useMemo } from 'react' +import { + Badge, + getCNPGPoolerMode, + getCNPGPoolerStatus, + getCNPGPoolerType, + isApiGroup, + type HealthLevel, +} from '@skyhook-io/k8s-ui' +import { + CNPGWorkspaceHeader, + CoverageNotice, + FilterChips, + Mono, + ScreenBody, + SectionTable, + Sub, + cnpgResource, + namespaceChip, + type CNPGScreenProps, +} from './shared' + +const SEVERITY: Record = { + healthy: 'success', + degraded: 'warning', + alert: 'alert', + unhealthy: 'error', + unknown: 'neutral', + neutral: 'neutral', +} + +export function CNPGPooling({ data, fleet, namespaces, searchParams, onSetParams, onInspect, inspected, onClearNamespaces }: CNPGScreenProps) { + const clusterFilter = searchParams.get('cluster') + const poolers = useMemo( + () => + (data.objects.poolers ?? []) + .filter((p) => isApiGroup(p.apiVersion, 'postgresql.cnpg.io')) + .filter((p) => !clusterFilter || `${p.metadata?.namespace}/${p.spec?.cluster?.name}` === clusterFilter), + [data.objects.poolers, clusterFilter], + ) + const chips = [ + ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), + ...namespaceChip(namespaces, onClearNamespaces), + ] + const readable = data.coverage.poolers?.state === 'full' || data.coverage.poolers?.state === 'partial' + + return ( +
+ + + + + <>{p.metadata?.name}{p.metadata?.namespace} }, + { + header: 'Target cluster', + width: '18%', + cell: (p) => { + const name = p.spec?.cluster?.name + const visible = fleet.rows.some((r) => r.namespace === p.metadata?.namespace && r.name === name) + return ( + <> + {name ?? '—'} + {name && !visible && not visible in this scope} + + ) + }, + }, + { header: 'Type', width: '8%', cell: (p) => {getCNPGPoolerType(p)} }, + { header: 'Mode', width: '12%', cell: (p) => getCNPGPoolerMode(p) }, + { + header: 'Instances', + width: '10%', + cell: (p) => ( + + {typeof p.status?.instances === 'number' ? p.status.instances : '–'}/{typeof p.spec?.instances === 'number' ? p.spec.instances : '–'} + + ), + }, + { + header: 'Status', + width: '14%', + cell: (p) => { + const st = getCNPGPoolerStatus(p) + return {st.text} + }, + }, + { + header: 'Connection pressure', + width: '16%', + cell: () => ( + <> + Not measured + Needs PgBouncer metrics + + ), + }, + ]} + rows={poolers} + rowKey={(p) => `${p.metadata?.namespace}/${p.metadata?.name}`} + rowResource={(p) => cnpgResource('poolers', p.metadata?.namespace, p.metadata?.name)} + onInspect={onInspect} + inspected={inspected} + minWidth={880} + empty={readable ? 'No Poolers in this scope.' : 'Poolers are not readable with your access.'} + footer="Instances are the Pooler’s own ready count. Client waits and server-pool saturation come from PgBouncer metrics, which Radar does not read yet." + /> + +
+ ) +} diff --git a/web/src/components/cnpg/CNPGProtection.tsx b/web/src/components/cnpg/CNPGProtection.tsx new file mode 100644 index 000000000..e354b12a6 --- /dev/null +++ b/web/src/components/cnpg/CNPGProtection.tsx @@ -0,0 +1,299 @@ +import { useMemo } from 'react' +import { + Badge, + FactValue, + cronToHuman, + formatAge, + formatDuration, + getCNPGBackupStatus, + getCNPGClusterBarmanPlugin, + getCNPGObjectStoreDestination, + getCNPGObjectStoreRecoveryWindows, + getCNPGScheduledBackupStatus, + isApiGroup, + toneTextClass, + type CNPGFleetRow, + type HealthLevel, +} from '@skyhook-io/k8s-ui' +import { + CNPGWorkspaceHeader, + CoverageNotice, + FilterChips, + Mono, + ScreenBody, + SectionTable, + Sub, + clusterResource, + cnpgResource, + namespaceChip, + type CNPGScreenProps, +} from './shared' + +const SEVERITY: Record = { + healthy: 'success', + degraded: 'warning', + alert: 'alert', + unhealthy: 'error', + unknown: 'neutral', + neutral: 'neutral', +} + +const WEEK_MS = 7 * 24 * 60 * 60 * 1000 + +function backupTime(b: any): string | undefined { + return b?.status?.startedAt || b?.metadata?.creationTimestamp +} + +function ageText(ts?: string): string { + return ts ? `${formatAge(ts)} ago` : '—' +} + +interface StoreRow { + key: string + namespace: string + name: string + destination: string + users: CNPGFleetRow[] + health: { text: string; tone: HealthLevel; evidence: string } +} + +function inferStoreHealth(store: any, users: CNPGFleetRow[]): StoreRow['health'] { + if (users.length === 0) { + return { text: 'Unknown', tone: 'unknown', evidence: 'No visible cluster uses this store' } + } + const failingArchiving = users.filter((u) => u.protection.walArchiving.tone === 'unhealthy') + const windows = getCNPGObjectStoreRecoveryWindows(store) + const failingBackups = windows.filter((w) => w.failingSinceLastSuccess) + if (failingArchiving.length > 0 || failingBackups.length > 0) { + const parts = [ + failingArchiving.length > 0 ? `WAL archiving failing on ${failingArchiving.map((u) => u.name).join(', ')}` : null, + failingBackups.length > 0 ? `a backup failed after the last success for ${failingBackups.map((w) => w.server).join(', ')}` : null, + ].filter(Boolean) + return { text: 'Uploads failing', tone: 'unhealthy', evidence: `Inferred: ${parts.join('; ')}` } + } + const archiving = users.filter((u) => u.protection.walArchiving.tone === 'healthy') + if (archiving.length > 0) { + return { text: 'Accepting uploads', tone: 'healthy', evidence: `Inferred from WAL archiving on ${archiving.map((u) => u.name).join(', ')}` } + } + return { text: 'Unknown', tone: 'unknown', evidence: 'Its clusters report no archiving result yet' } +} + +export function CNPGProtection({ data, fleet, namespaces, searchParams, onSetParams, onInspect, inspected, onClearNamespaces }: CNPGScreenProps) { + const clusterFilter = searchParams.get('cluster') + const rows = useMemo( + () => fleet.rows.filter((r) => !clusterFilter || `${r.namespace}/${r.name}` === clusterFilter), + [fleet.rows, clusterFilter], + ) + + const failed = useMemo(() => { + const now = Date.now() + return (data.objects.backups ?? []) + .filter((b) => isApiGroup(b.apiVersion, 'postgresql.cnpg.io')) + .filter((b) => { + const level = getCNPGBackupStatus(b).level + if (level !== 'unhealthy' && level !== 'alert') return false + const t = Date.parse(backupTime(b) ?? '') + return Number.isFinite(t) && now - t <= WEEK_MS + }) + .filter((b) => !clusterFilter || `${b.metadata?.namespace}/${b.spec?.cluster?.name}` === clusterFilter) + .sort((a, b) => Date.parse(backupTime(b) ?? '') - Date.parse(backupTime(a) ?? '')) + }, [data.objects.backups, clusterFilter]) + + const stores = useMemo(() => { + return (data.objects.objectStores ?? []).map((s) => { + const ns = s.metadata?.namespace ?? '' + const name = s.metadata?.name ?? '' + const users = fleet.rows.filter( + (r) => r.namespace === ns && getCNPGClusterBarmanPlugin(r.cluster)?.barmanObjectName === name, + ) + return { key: `${ns}/${name}`, namespace: ns, name, destination: getCNPGObjectStoreDestination(s), users, health: inferStoreHealth(s, users) } + }).filter((s) => !clusterFilter || s.users.some((u) => `${u.namespace}/${u.name}` === clusterFilter)) + }, [data.objects.objectStores, fleet.rows, clusterFilter]) + + const schedules = useMemo( + () => + (data.objects.scheduledBackups ?? []).filter( + (s) => !clusterFilter || `${s.metadata?.namespace}/${s.spec?.cluster?.name}` === clusterFilter, + ), + [data.objects.scheduledBackups, clusterFilter], + ) + + const chips = [ + ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), + ...namespaceChip(namespaces, onClearNamespaces), + ] + const backupsReadable = data.coverage.backups?.state === 'full' || data.coverage.backups?.state === 'partial' + + return ( +
+ + + + + + ( + <> +
{r.name}
+ {r.namespace} + + ), + }, + { header: 'Schedule', width: '13%', cell: (r) => }, + { + header: 'Last successful backup', + width: '16%', + cell: (r) => ( + <> + + {r.protection.lastSuccessfulBackup.source && {r.protection.lastSuccessfulBackup.source}} + + ), + }, + { header: 'WAL archiving', width: '16%', cell: (r) => }, + { + header: 'Recovery window', + width: '13%', + cell: (r) => + r.protection.recoveryWindow.from ? ( + <> + + from {ageText(r.protection.recoveryWindow.from)} + + + to {r.protection.recoveryWindow.to ? ageText(r.protection.recoveryWindow.to) : 'unknown'} + {r.protection.recoveryWindow.tone === 'degraded' ? ' · not advancing' : ''} + + + ) : ( + + ), + }, + { + header: 'Restore validation', + width: '13%', + cell: (r) => ( + + + + ), + }, + { + header: 'Destination', + width: '15%', + cell: (r) => , + }, + ]} + rows={rows} + rowKey={(r) => r.key} + rowResource={(r) => clusterResource(r.namespace, r.name)} + onInspect={onInspect} + inspected={inspected} + minWidth={1000} + empty="No PostgreSQL clusters in this scope." + footer="Kubernetes records no restore tests, so restore validation is never shown as passed. Recovery windows come from ObjectStore status." + /> + + {b.metadata?.name} }, + { header: 'Cluster', width: '16%', cell: (b) => <>{b.spec?.cluster?.name ?? '—'}{b.metadata?.namespace} }, + { header: 'Started', width: '12%', cell: (b) => ageText(backupTime(b)) }, + { + header: 'Error', + width: '44%', + cell: (b) => {b.status?.error || getCNPGBackupStatus(b).text}, + }, + ]} + rows={failed} + rowKey={(b) => `${b.metadata?.namespace}/${b.metadata?.name}`} + rowResource={(b) => cnpgResource('backups', b.metadata?.namespace, b.metadata?.name)} + onInspect={onInspect} + inspected={inspected} + empty={backupsReadable ? 'No failed backups in the last 7 days.' : 'Backups are not readable with your access.'} + /> + + <>{s.name}{s.namespace} }, + { header: 'Destination', width: '30%', cell: (s) => {s.destination} }, + { header: 'Used by', width: '20%', cell: (s) => (s.users.length ? s.users.map((u) => u.name).join(', ') : None visible) }, + { + header: 'Upload health (inferred)', + width: '32%', + cell: (s) => ( + <> + {s.health.text} + {s.health.evidence} + + ), + }, + ]} + rows={stores} + rowKey={(s) => s.key} + rowResource={(s) => cnpgResource('objectstores', s.namespace, s.name, 'barmancloud.cnpg.io')} + onInspect={onInspect} + inspected={inspected} + empty={data.coverage.objectStores?.state === 'notInstalled' ? 'The barman-cloud plugin’s ObjectStore kind is not installed.' : 'No ObjectStores in this scope.'} + footer="ObjectStore has no health status of its own; upload health is inferred from its clusters’ WAL archiving and backup results." + /> + + <>{s.metadata?.name}{s.metadata?.namespace} }, + { header: 'Cluster', width: '16%', cell: (s) => s.spec?.cluster?.name ?? '—' }, + { + header: 'Schedule', + width: '24%', + cell: (s) => ( + <> + {s.spec?.schedule ?? '—'} + {s.spec?.schedule && {cronToHuman(s.spec.schedule)}} + + ), + }, + { + header: 'Status', + width: '12%', + cell: (s) => { + const st = getCNPGScheduledBackupStatus(s) + return {st.text} + }, + }, + { header: 'Last run', width: '12%', cell: (s) => ageText(s.status?.lastScheduleTime) }, + { + header: 'Next run', + width: '12%', + cell: (s) => { + const next = s.status?.nextScheduleTime + if (!next) return '—' + const ms = Date.parse(next) - Date.now() + return ms >= 0 ? `in ${formatDuration(ms)}` : `${formatDuration(-ms)} overdue` + }, + }, + ]} + rows={schedules} + rowKey={(s) => `${s.metadata?.namespace}/${s.metadata?.name}`} + rowResource={(s) => cnpgResource('scheduledbackups', s.metadata?.namespace, s.metadata?.name)} + onInspect={onInspect} + inspected={inspected} + empty={data.coverage.scheduledBackups?.state === 'full' ? 'No ScheduledBackups in this scope.' : 'ScheduledBackups are not fully readable with your access.'} + footer={data.backupsOmitted > 0 ? `${data.backupsOmitted} settled backups older than 7 days are not listed.` : undefined} + /> +
+
+ ) +} diff --git a/web/src/components/cnpg/CNPGSummaryHost.tsx b/web/src/components/cnpg/CNPGSummaryHost.tsx index 10f9d3b39..bb3dab376 100644 --- a/web/src/components/cnpg/CNPGSummaryHost.tsx +++ b/web/src/components/cnpg/CNPGSummaryHost.tsx @@ -1,8 +1,30 @@ import type { ReactNode } from 'react' -import { useNavigate } from 'react-router-dom' -import { CNPGClusterSummary, PaneLoader, refToSelectedResource, type CNPGRef, type NavigateToResource } from '@skyhook-io/k8s-ui' +import { useLocation, useNavigate, useSearchParams } from 'react-router-dom' +import { ArrowLeft } from 'lucide-react' +import { + CNPG_BARMAN_OBJECTSTORE_GROUP, + CNPG_GROUP, + CNPG_KIND_BY_KEY, + CNPGBackupSummary, + CNPGClusterSummary, + CNPGDatabaseSummary, + CNPGImageCatalogSummary, + CNPGObjectStoreSummary, + CNPGPoolerSummary, + CNPGPublicationSummary, + CNPGScheduledBackupSummary, + CNPGSubscriptionSummary, + PaneLoader, + isApiGroup, + refToSelectedResource, + type CNPGNavigate, + type CNPGRef, + type CNPGWorkspaceResponse, + type NavigateToResource, +} from '@skyhook-io/k8s-ui' import { useCNPGFleet } from './useCNPGSidebarWorkspace' import { cnpgClusterFullPath } from './paths' +import { decodeDrawerTrail, encodeDrawerTrail } from './routes' interface SummaryContext { apiKind: string @@ -41,12 +63,88 @@ function ClusterSummaryHost({ namespace, name, context, onNavigate }: SummaryCon ) } +type ObjectSummary = (props: { resource: any; workspace: CNPGWorkspaceResponse | null; onNavigate?: CNPGNavigate }) => ReactNode + +const OBJECT_SUMMARIES: Record = { + Backup: CNPGBackupSummary, + ScheduledBackup: CNPGScheduledBackupSummary, + Pooler: CNPGPoolerSummary, + Database: CNPGDatabaseSummary, + Publication: CNPGPublicationSummary, + Subscription: CNPGSubscriptionSummary, + ImageCatalog: CNPGImageCatalogSummary, + ClusterImageCatalog: CNPGImageCatalogSummary, +} + +function ObjectSummaryHost({ ctx, Summary }: { ctx: SummaryContext; Summary: ObjectSummary }) { + // A ClusterImageCatalog is referenced from any namespace, so its users are + // read across every namespace the caller can see. + const clusterScoped = ctx.resource?.kind === 'ClusterImageCatalog' + const { query } = useCNPGFleet(clusterScoped ? [] : [ctx.namespace]) + if (query.isLoading) return + const workspace = query.data?.installed ? query.data : null + const go = ctx.onNavigate ? (ref: CNPGRef) => ctx.onNavigate?.(refToSelectedResource(ref)) : undefined + return +} + +// The object's own apiVersion decides: Velero also ships a Backup kind. +function groupOf(ctx: SummaryContext): string | undefined { + const apiVersion = ctx.resource?.apiVersion + if (typeof apiVersion !== 'string') return ctx.group + if (isApiGroup(apiVersion, CNPG_GROUP)) return CNPG_GROUP + if (isApiGroup(apiVersion, CNPG_BARMAN_OBJECTSTORE_GROUP)) return CNPG_BARMAN_OBJECTSTORE_GROUP + return undefined +} + +function renderSummaryFor(ctx: SummaryContext): ReactNode { + const group = groupOf(ctx) + const kind = ctx.resource?.kind + if (group === CNPG_BARMAN_OBJECTSTORE_GROUP && kind === 'ObjectStore') { + return + } + if (group !== CNPG_GROUP) return null + if (kind === 'Cluster') return + const Summary = OBJECT_SUMMARIES[kind] + return Summary ? : null +} + +const KIND_BY_PLURAL: Record = Object.fromEntries( + Object.values(CNPG_KIND_BY_KEY).map((k) => [k.plural, k.kind]), +) + +// On a workspace screen the drawer URL carries the chain of objects opened +// from inside it; this renders the step back to the previous one. +function DrawerTrailBack({ name, children }: { name: string; children: ReactNode }) { + const location = useLocation() + const [searchParams, setSearchParams] = useSearchParams() + const trail = decodeDrawerTrail(searchParams.get('drawer')) + const current = trail[trail.length - 1] + if (!location.pathname.startsWith('/cnpg') || trail.length < 2 || current?.name !== name) return <>{children} + const prev = trail[trail.length - 2] + const back = () => { + const params = new URLSearchParams(searchParams) + params.set('drawer', encodeDrawerTrail(trail.slice(0, -1))) + setSearchParams(params, { replace: true }) + } + return ( + <> +
+ +
+ {children} + + ) +} + /** * The composed Overview for CloudNativePG kinds. Returns null for kinds * without one, which keeps the default Overview. */ export function renderCNPGSummary(ctx: SummaryContext): ReactNode { - if (ctx.group !== 'postgresql.cnpg.io') return null - if (ctx.resource?.kind === 'Cluster') return - return null + const node = renderSummaryFor(ctx) + if (!node || ctx.context !== 'drawer') return node + return {node} } diff --git a/web/src/components/cnpg/CNPGView.tsx b/web/src/components/cnpg/CNPGView.tsx index f6ba9c890..91b99105b 100644 --- a/web/src/components/cnpg/CNPGView.tsx +++ b/web/src/components/cnpg/CNPGView.tsx @@ -6,6 +6,11 @@ import { useAPIResources } from '../../api/apiResources' import { usePinnedKinds } from '../../hooks/useFavorites' import { useResourceCounts } from '../../hooks/useResourceCounts' import { CNPGOverview } from './CNPGOverview' +import { CNPGProtection } from './CNPGProtection' +import { CNPGDeclarations } from './CNPGDeclarations' +import { CNPGPooling } from './CNPGPooling' +import { CNPGOperator } from './CNPGOperator' +import { CNPGScreenGate } from './shared' import { decodeDrawerTrail, encodeDrawerTrail, parseCNPGRoute, sameResource } from './routes' import { useCNPGFleet, useCNPGSidebarWorkspace } from './useCNPGSidebarWorkspace' @@ -110,16 +115,32 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose categoryWorkspaces={sidebarWorkspace} />
- + + {(data, readyFleet) => { + const props = { + data, + fleet: readyFleet, + namespaces, + searchParams, + onSetParams: setParams, + onInspect: inspect, + inspected: drawerTarget, + onClearNamespaces, + } + switch (route.screen) { + case 'protection': + return + case 'declarations': + return + case 'pooling': + return + case 'operator': + return + default: + return + } + }} +
) diff --git a/web/src/components/cnpg/routes.test.ts b/web/src/components/cnpg/routes.test.ts index fb19f5209..15f5a9667 100644 --- a/web/src/components/cnpg/routes.test.ts +++ b/web/src/components/cnpg/routes.test.ts @@ -6,6 +6,8 @@ describe('CNPG routes', () => { expect(parseCNPGRoute('/cnpg').screen).toBe('overview') expect(parseCNPGRoute('/cnpg/').screen).toBe('overview') expect(parseCNPGRoute('/cnpg/nope').screen).toBe('overview') + expect(parseCNPGRoute('/cnpg/protection').screen).toBe('protection') + expect(parseCNPGRoute('/cnpg/operator').screen).toBe('operator') }) it('round-trips a drawer trail and keeps the API group', () => { diff --git a/web/src/components/cnpg/routes.ts b/web/src/components/cnpg/routes.ts index e4d4b609a..4621ee9e0 100644 --- a/web/src/components/cnpg/routes.ts +++ b/web/src/components/cnpg/routes.ts @@ -10,9 +10,6 @@ export const CNPG_SCREENS: { id: CNPGScreen; label: string; path: string }[] = [ { id: 'operator', label: 'Operator', path: '/cnpg/operator' }, ] -/** Screens that exist in this build. Destinations appear in the sidebar only once their screen does. */ -export const CNPG_AVAILABLE_SCREENS: ReadonlySet = new Set(['overview']) - export interface CNPGRoute { screen: CNPGScreen } @@ -22,7 +19,7 @@ export function parseCNPGRoute(pathname: string): CNPGRoute { if (seg[0] !== 'cnpg') return { screen: 'overview' } const s = seg[1] ?? '' const match = CNPG_SCREENS.find((x) => x.id === s) - if (match && CNPG_AVAILABLE_SCREENS.has(match.id)) return { screen: match.id } + if (match) return { screen: match.id } return { screen: 'overview' } } diff --git a/web/src/components/cnpg/shared.tsx b/web/src/components/cnpg/shared.tsx new file mode 100644 index 000000000..740d96da1 --- /dev/null +++ b/web/src/components/cnpg/shared.tsx @@ -0,0 +1,265 @@ +import type { ReactNode } from 'react' +import type { UseQueryResult } from '@tanstack/react-query' +import { clsx } from 'clsx' +import { Database, X } from 'lucide-react' +import { + CNPG_KIND_BY_KEY, + PaneLoader, + type CNPGFleet, + type CNPGWorkspaceResponse, +} from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { useConnection } from '../../context/ConnectionContext' +import { EmptyState, Notice, ROW_HOVER, TABLE_HEAD, TABLE_WRAP, TBODY, TD, TH } from '../capacity/shared' +import { sameResource } from './routes' + +export interface CNPGScreenProps { + data: CNPGWorkspaceResponse + fleet: CNPGFleet + namespaces: string[] + searchParams: URLSearchParams + onSetParams: (update: Record) => void + onInspect: (resource: SelectedResource) => void + inspected: SelectedResource | null + onClearNamespaces: () => void +} + +const COVERAGE_LABEL: Record = { + denied: 'no access', + partial: 'no access in some namespaces', + syncing: 'still loading', + error: 'could not be read', +} + +export function CNPGWorkspaceHeader({ title, subtitle, actions }: { title: string; subtitle?: ReactNode; actions?: ReactNode }) { + return ( +
+
+ + CloudNativePG +
+
+
+

{title}

+ {subtitle &&
{subtitle}
} +
+ {actions} +
+
+ ) +} + +export function CoverageNotice({ fleet, data }: { fleet: CNPGFleet; data: CNPGWorkspaceResponse }) { + if (fleet.incompleteKinds.length === 0) return null + const parts = fleet.incompleteKinds.map((k) => { + const cov = data.coverage[k] + const label = COVERAGE_LABEL[cov?.state ?? ''] ?? cov?.state + return `${CNPG_KIND_BY_KEY[k].kind} (${label})` + }) + return ( + + Some CloudNativePG data is not readable: {parts.join(', ')}. Facts built on it read “No access” or “unknown” rather than none, and counts are lower bounds. + + ) +} + +/** + * Loading, error and not-installed states every workspace screen shares. + * Renders the screen only once there is workspace data to render. + */ +export function CNPGScreenGate({ + query, + fleet, + children, +}: { + query: UseQueryResult + fleet: CNPGFleet | null + children: (data: CNPGWorkspaceResponse, fleet: CNPGFleet) => ReactNode +}) { + const { connection } = useConnection() + const data = query.data + if (!data && query.isLoading) return + if (!data) { + return ( + + ) + } + if (!data.installed || !fleet) { + return ( + + ) + } + return <>{children(data, fleet)} +} + +export function ScreenBody({ children }: { children: ReactNode }) { + return ( +
+
{children}
+
+ ) +} + +export function FilterChips({ chips }: { chips: { label: string; onClear: () => void }[] }) { + if (chips.length === 0) return null + return ( +
+ {chips.map((c) => ( + + {c.label} + + + ))} +
+ ) +} + +export function namespaceChip(namespaces: string[], onClear: () => void) { + return namespaces.length > 0 ? [{ label: `Namespace: ${namespaces.join(', ')}`, onClear }] : [] +} + +export function Segments({ + value, + options, + onChange, + label, +}: { + value: T + options: { id: T; label: string; count?: number }[] + onChange: (id: T) => void + label: string +}) { + return ( +
+ {options.map((o) => { + const on = o.id === value + return ( + + ) + })} +
+ ) +} + +export interface TableColumn { + header: ReactNode + width?: string + cell: (row: T) => ReactNode + className?: string +} + +/** A workspace table. Rows inspect in the drawer; the inspected row is highlighted. */ +export function SectionTable({ + title, + subtitle, + columns, + rows, + rowKey, + rowResource, + onInspect, + inspected, + empty, + minWidth = 760, + footer, +}: { + title: ReactNode + subtitle?: ReactNode + columns: TableColumn[] + rows: T[] + rowKey: (row: T) => string + rowResource?: (row: T) => SelectedResource | null + onInspect?: (resource: SelectedResource) => void + inspected?: SelectedResource | null + empty: ReactNode + minWidth?: number + footer?: ReactNode +}) { + return ( +
+
+

{title}

+ {subtitle && {subtitle}} +
+
+ {rows.length === 0 ? ( +
{empty}
+ ) : ( +
+ + + {columns.map((c, i) => ( + + ))} + + + + {columns.map((c, i) => ( + + ))} + + + + {rows.map((row) => { + const res = rowResource?.(row) ?? null + const active = !!res && sameResource(inspected, res) + return ( + onInspect(res) : undefined} + className={clsx(res && onInspect && 'cursor-pointer', ROW_HOVER, active && 'selection')} + aria-selected={res ? active : undefined} + > + {columns.map((c, i) => ( + + ))} + + ) + })} + +
{c.header}
{c.cell(row)}
+
+ )} +
+ {footer &&
{footer}
} +
+ ) +} + +export function Mono({ children, title }: { children: ReactNode; title?: string }) { + return {children} +} + +export function Sub({ children }: { children: ReactNode }) { + return
{children}
+} + +export function clusterResource(namespace: string, name: string): SelectedResource { + return { kind: 'clusters', group: 'postgresql.cnpg.io', namespace, name } +} + +export function cnpgResource(plural: string, namespace: string, name: string, group = 'postgresql.cnpg.io'): SelectedResource { + return { kind: plural, group, namespace, name } +} diff --git a/web/src/components/cnpg/useCNPGSidebarWorkspace.ts b/web/src/components/cnpg/useCNPGSidebarWorkspace.ts index fbf7568fd..d963a5698 100644 --- a/web/src/components/cnpg/useCNPGSidebarWorkspace.ts +++ b/web/src/components/cnpg/useCNPGSidebarWorkspace.ts @@ -4,7 +4,7 @@ import { Database, FileCheck2, Settings2, ShieldCheck, Waypoints } from 'lucide- import { buildCNPGFleet, type CNPGFleet, type SidebarCategoryWorkspace } from '@skyhook-io/k8s-ui' import type { APIResource } from '../../types' import { useCNPGWorkspace } from '../../api/cnpg' -import { CNPG_AVAILABLE_SCREENS, CNPG_SCREENS, type CNPGScreen } from './routes' +import { CNPG_SCREENS, type CNPGScreen } from './routes' export const CNPG_SIDEBAR_CATEGORY = 'CloudNativePG' @@ -64,7 +64,7 @@ export function useCNPGSidebarWorkspace({ return useMemo(() => { if (!discovered) return undefined - const destinations = CNPG_SCREENS.filter((s) => CNPG_AVAILABLE_SCREENS.has(s.id)).map((s) => { + const destinations = CNPG_SCREENS.map((s) => { const { count, title } = destinationCount(s.id, fleet) return { id: s.id, From b698092048a4fed5a3266b21d9aa59805fdb29dd Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Tue, 29 Sep 2026 02:46:45 +0300 Subject: [PATCH 03/17] Add the CloudNativePG full detail, merged instance logs and cluster Activity - Every CNPG kind now has a CNPG-framed full detail at /cnpg///, reached from row Open, drawer expand and a redirect from /workload/...: the workspace sidebar stays highlighted with the object nested under its home, a return link names the page the drilldown came from (history state, kept across tab changes), and a crumb names the object's place. - Detail URLs pin the kube context (ctx=, added on first view). After a context switch the page says the object is not in the new context and offers Switch back / Go to Overview instead of loading a same-named object. - A Cluster adds Protection (its recovery evidence), Activity (replacing Timeline) and merged instance Logs; WorkloadView gains additive extraTabs and the logs viewer an initialPods selection. - GET /api/cnpg/clusters/{ns}/{name}/logs (+/stream): logs from controller-owned instance Pods, gated on get clusters, list pods and get pods/log, with CNPG's JSON lines parsed into level/logger/message. - GET /api/cnpg/clusters/{ns}/{name}/activity: timeline events for the Cluster, its instance Pods and CNPG objects attributed to it, per-kind gated. The timeline now records the owning cluster (cnpg.io/cluster label or spec.cluster.name) at ingestion so deleted Backups and declarations stay attributed; earlier history is reported as incomplete. - The drawer trail back link moves to the drawer shell so it survives hops to Pods and Secrets. Review fixes on the workspace screens and summaries: empty states follow each kind's coverage, Declarations separates not applied from pending, GitOps provenance reads "not recorded" instead of "applied directly", ObjectStore no longer asserts recoverability, a Backup's destination inferred from the Cluster's current config is labelled, schedule runs require the owner UID, partial catalog coverage keeps the known users, and CNPG's six-field cron is shown verbatim. Docs: docs/cnpg.md (destinations, navigation, certainty table, access). --- CLAUDE.md | 3 + README.md | 2 +- docs/cnpg.md | 66 +++ docs/integrations.md | 2 + internal/server/cnpg_cluster_activity.go | 228 +++++++++ internal/server/cnpg_cluster_history_test.go | 431 ++++++++++++++++ internal/server/cnpg_cluster_logs.go | 465 ++++++++++++++++++ internal/server/server.go | 3 + internal/server/workload_logs.go | 6 +- .../src/components/cnpg/CNPGBackupSummary.tsx | 20 +- .../cnpg/CNPGDeclarativeSummary.tsx | 2 +- .../cnpg/CNPGImageCatalogSummary.tsx | 8 +- .../cnpg/CNPGObjectStoreSummary.tsx | 24 +- .../cnpg/CNPGObjectSummary.test.tsx | 44 +- .../src/components/cnpg/relations.test.ts | 39 +- .../k8s-ui/src/components/cnpg/relations.ts | 61 ++- .../components/logs/WorkloadLogsViewer.tsx | 14 +- .../src/components/workload/WorkloadView.tsx | 34 +- .../k8s-ui/src/components/workload/index.ts | 1 + pkg/timeline/converter.go | 36 +- pkg/timeline/converter_cnpg_test.go | 66 +++ web/src/App.tsx | 23 +- web/src/api/cnpg.ts | 28 +- .../components/cnpg/CNPGClusterActivity.tsx | 48 ++ web/src/components/cnpg/CNPGClusterLogs.tsx | 59 +++ web/src/components/cnpg/CNPGDeclarations.tsx | 50 +- web/src/components/cnpg/CNPGDetailPage.tsx | 228 +++++++++ web/src/components/cnpg/CNPGDrawerTrail.tsx | 37 ++ web/src/components/cnpg/CNPGOperator.tsx | 6 +- web/src/components/cnpg/CNPGOverview.tsx | 48 +- web/src/components/cnpg/CNPGPooling.tsx | 4 +- web/src/components/cnpg/CNPGProtection.tsx | 42 +- web/src/components/cnpg/CNPGSummaryHost.tsx | 73 ++- web/src/components/cnpg/CNPGView.tsx | 17 +- web/src/components/cnpg/paths.ts | 14 +- web/src/components/cnpg/routes.test.ts | 26 +- web/src/components/cnpg/routes.ts | 53 +- web/src/components/cnpg/shared.tsx | 25 + .../resources/ResourceDetailDrawer.tsx | 6 + web/src/components/workload/WorkloadView.tsx | 24 +- 40 files changed, 2193 insertions(+), 173 deletions(-) create mode 100644 docs/cnpg.md create mode 100644 internal/server/cnpg_cluster_activity.go create mode 100644 internal/server/cnpg_cluster_history_test.go create mode 100644 internal/server/cnpg_cluster_logs.go create mode 100644 pkg/timeline/converter_cnpg_test.go create mode 100644 web/src/components/cnpg/CNPGClusterActivity.tsx create mode 100644 web/src/components/cnpg/CNPGClusterLogs.tsx create mode 100644 web/src/components/cnpg/CNPGDetailPage.tsx create mode 100644 web/src/components/cnpg/CNPGDrawerTrail.tsx diff --git a/CLAUDE.md b/CLAUDE.md index 48d40150b..c769a06e4 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -27,6 +27,7 @@ Not everything is in this file. The following files contain critical details tha | Adding or modifying **HTTP endpoints** | `internal/server/server.go` — all routes are defined here | | Adding or modifying **CLI flags** | `cmd/explorer/main.go` — flag definitions and defaults | | Adding a **new CRD integration** (renderer, topology, discovery) | [docs/INTEGRATION_GUIDE.md](docs/INTEGRATION_GUIDE.md) — full checklist with collision gotchas | +| Working on the **CloudNativePG workspace** (`/cnpg`) | [docs/cnpg.md](docs/cnpg.md) — destinations, navigation (drawer trail, return label, `ctx` guard) and the certainty table: which source each fact comes from and what it reads when unknown. Data from `/api/cnpg/workspace` (per-kind coverage); derivations in `packages/k8s-ui/src/components/cnpg/workspace.ts` + `relations.ts`; screens in `web/src/components/cnpg/` | | Working on the **Capacity (Karpenter) views** | [docs/capacity.md](docs/capacity.md) — the four screens, the per-value certainty contract (`= ≥ ≤ ?`, unavailable ≠ zero, partial ≠ exact, declared ≠ actual), demand-evaluation semantics, and the real Karpenter failure model. Wire types in `pkg/capacityapi`, engine in `internal/capacity`, handlers in `internal/server/capacity*` | | Working on **resource renderers** | `packages/k8s-ui/src/components/resources/renderers/` — all existing renderers live here | | Understanding **cluster connection behavior** | [docs/configuration.md](docs/configuration.md) — kubeconfig precedence, multi-context, in-cluster | @@ -185,6 +186,8 @@ After `make -demo`, run `kubectl config use-context kind-radar--demo - CloudNativePG workspace: `/api/cnpg/workspace` returns every CNPG kind (plus owner-validated instance Pods) with per-kind `coverage` (`full|partial|denied|notInstalled|syncing|error`, `partial` naming only in-scope denied namespaces), CNPG issues from the Issues engine and `cnpgNoDeclarativeBackup` audit findings, each withheld where the caller lacks coverage. Namespaced kinds follow the view filter and the capacity per-namespace `list` fallback; `ClusterImageCatalog` needs a cluster-scope `list`. Handler `internal/server/cnpg_workspace.go` - CloudNativePG operator: `/api/cnpg/operator` returns the operator Deployments (`app.kubernetes.io/name=cloudnative-pg`) and plugin Deployments (served by Services labelled `cnpg.io/pluginName`) with image-tag version and readiness (`null` when unreported), plus the operator's ConfigMap/Secret/monitoring-queries references from its args and env. ConfigMap data only with `get configmaps`; the Secret is name-only, never read. Deployments and Services carry the workspace `coverage` states; deliberately ignores the namespace view filter (the operator lives in its own namespace). Handler `internal/server/cnpg_operator.go` - CloudNativePG reverse-lookup: `/api/cnpg/imagecatalogs/{ns}/{name}/clusters` and `/api/cnpg/clusterimagecatalogs/{name}/clusters` return the Clusters pinned to an image catalog, with the major each asks for and the image it actually resolved. Cluster-scoped catalogs are referenceable from any namespace, so the cluster-scoped route reads cluster-wide gated on `list clusters` — a view-filtered answer would report "nothing uses this" before an edit. +- CloudNativePG Cluster logs: `/api/cnpg/clusters/{ns}/{name}/logs` (bounded snapshot) and `/logs/stream` (SSE, re-resolves instances every 5s) merge every instance Pod — label `cnpg.io/cluster` AND controller ownerRef to the Cluster's UID, never the label alone. Gated on `get clusters` + `list pods` + `get pods/log` before the Cluster lookup (404 after). `container` defaults to `postgres`, `tailLines` 200, `sinceTime` is converted to seconds and trimmed, `pod` must be a validated instance (400). Entries keep raw `content` and add `level`/`logger`/`message` parsed from the instance manager's JSON (`record.error_severity` wins over `level`). +- CloudNativePG Cluster activity: `/api/cnpg/clusters/{ns}/{name}/activity?since=&limit=` reads the timeline store for the Cluster, its instance Pods (by owner) and CNPG children attributed by the retained `cnpg.io/cluster` label — `pkg/timeline.ExtractLabels` records it from the label or `spec.cluster.name` on CNPG-group objects and Pods, so deleted children stay attributed. K8s Event rows join by subject UID. Rows of a kind the caller can't `list` in the namespace are dropped; `oldest` is the namespace's retention floor and `attributionSince` the earliest labelled row — history before it cannot attribute deleted children. ## Key Patterns diff --git a/README.md b/README.md index d13c7f0c3..f661e6414 100644 --- a/README.md +++ b/README.md @@ -546,7 +546,7 @@ Upgrade impact also gets list-only access to CSIStorageCapacities, FlowSchemas, | **Strimzi** | [KafkaConnector failure evidence](docs/integrations.md#strimzi-kafka-connectors) (connector/task status) | | **Velero** | Backup, Restore, Schedule, BackupStorageLocation, VolumeSnapshotLocation | | **External Secrets** | ExternalSecret, ClusterExternalSecret, SecretStore, ClusterSecretStore | -| **CloudNativePG** | Cluster, Backup, ScheduledBackup, Pooler | +| **CloudNativePG** | Cluster, Backup, ScheduledBackup, Pooler, Database, Publication, Subscription, ImageCatalog, ClusterImageCatalog, ObjectStore — plus a [workspace](docs/cnpg.md) for fleet, protection and declaration triage | | **Crossplane** | Managed Resources (any provider), Composite Resources, Claims, Provider, ProviderConfig, Function, Configuration, Composition, CompositionRevision, XRD | | **Kyverno** | Policy, ClusterPolicy, PolicyReport, ClusterPolicyReport | | **Sealed Secrets** | SealedSecret | diff --git a/docs/cnpg.md b/docs/cnpg.md new file mode 100644 index 000000000..ce7ce74cb --- /dev/null +++ b/docs/cnpg.md @@ -0,0 +1,66 @@ +# CloudNativePG workspace + +A task-shaped view over [CloudNativePG](https://cloudnative-pg.io/) (CNPG): which PostgreSQL cluster needs attention, why, and what to inspect next — without assembling the story from ten separate CRD lists. The per-kind renderers, issue detection and audit check it builds on are described in [integrations.md](integrations.md#cloudnativepg). + +The workspace is read-only. It never writes to a cluster. + +## Where it lives + +CNPG stays inside **Resources**; there is no new global navigation item. When the `postgresql.cnpg.io` CRDs are discovered, the Resources sidebar's CloudNativePG group gains a **Workspace** block above its exact kinds: + +| Destination | Route | Job | Detail home for | +|---|---|---|---| +| Overview | `/cnpg` | The fleet: every Cluster with instances, replication, protection, declarations and its top problem. Defaults to **Needs attention**. | Cluster | +| Protection | `/cnpg/protection` | Recovery evidence per cluster, failed backups (7 days), destinations, schedules. | Backup, ScheduledBackup, ObjectStore | +| Declarations | `/cnpg/declarations` | Databases, Publications, Subscriptions and managed roles by cluster; declared vs reconciled. | Database, Publication, Subscription | +| Pooling | `/cnpg/pooling` | Poolers and the clusters they front. | Pooler | +| Operator | `/cnpg/operator` | Operator and plugin workloads, image catalogs, operator configuration. | ImageCatalog, ClusterImageCatalog | + +Destination badges count **affected clusters**, not findings, and follow the namespace filter (the sidebar says so). The exact kinds stay under a collapsible **Resource kinds** block, grouped by API group; on workspace screens it starts collapsed. + +Every CNPG kind's full detail is `/cnpg///` — reached from a row's **Open**, from the drawer's expand control, and by redirect from the generic `/workload/...` URL. The page keeps the workspace sidebar (its destination highlighted, the object nested under it) and uses Radar's detail view underneath: **Overview** is a composed summary, **Spec & status** is the kind's existing renderer, then YAML and the rest. A Cluster adds **Protection** (its recovery evidence), **Activity** (in place of Timeline) and merged instance **Logs**. + +## Navigation + +- **One drawer.** Rows inspect in the app's single drawer; `?drawer=kind:group:namespace:name` backs it, so refresh, share and Back restore it. Links inside the drawer append to that chain and show "← " at the top of the drawer. +- **Return vs location.** A full detail shows "← " only when it was reached by a drilldown (the label travels in history state); sidebar and global-nav hops are location changes and carry no return label. The crumb (`CloudNativePG / Protection / name`) always names the object's place, so a fresh tab has a parent without a fabricated previous task. +- **Context.** Detail URLs carry `ctx=` (added on first view when absent). After a context switch the page says " is not in " with **Switch back** and **Go to …** — Radar never opens a same-named object from another cluster. +- **Namespace filter** narrows collections and counts. An explicitly opened object stays open, with a note when it is outside the filter. + +## The certainty contract + +Every value is something the cluster reports, labelled with where it came from. When the cluster does not report something the UI says so; it never shows zero, "none" or green in its place. + +| Fact | Source | When it is not known | +|---|---|---| +| Instances, primary | `status.readyInstances`, `status.currentPrimary`, instance Pods (controller-owned by the Cluster's UID) | `–` | +| Replication | Pod readiness only | Always "lag unknown": readiness does not show whether a replica is streaming. Lag needs runtime data Radar does not read yet. | +| Schedule | ScheduledBackups targeting the Cluster (`spec.suspend` → suspended) | "No access to ScheduledBackups" when unreadable in that namespace | +| Destination | barman-cloud plugin `barmanObjectName`, in-tree `barmanObjectStore`, or volume snapshots | "No destination configured" | +| Last successful backup | Newest of: completed Backup CRs (7-day window plus the newest per cluster), ObjectStore `serverRecoveryWindow[...].lastSuccessfulBackupTime`, in-tree `status.lastSuccessfulBackup` (ignored for plugin clusters, where CNPG no longer sets it) — the winning source is shown | "None observed", or "No access to Backups" | +| WAL archiving | `ContinuousArchiving` condition | "Not reported" | +| Recovery window | ObjectStore `status.serverRecoveryWindow` for the cluster's server name | "Not reported" | +| Restore validation | A Cluster in the same namespace bootstrapped (`bootstrap.recovery`) from this cluster's store/server or one of its Backups, **with a ready instance** | "None recorded" (unknown tone) — Kubernetes records no restore tests, so this is never green. A matching cluster without a ready instance reads "Recovery declared in …". | +| ObjectStore upload health | **Inferred** from its user clusters' WAL archiving and recovery windows (ObjectStore has no status of its own) | "Unknown" | +| Declarations | `status.applied` (true / false / absent = pending); managed roles from `status.managedRolesStatus` (`reconciled`, `cannotReconcile`; anything else pending) | Pending, never failed | +| GitOps source | Argo CD / Flux labels and the Argo tracking annotation | "GitOps source not recorded" | +| Pooler pressure | — | "Not measured": needs PgBouncer metrics | +| ScheduledBackup cron | Shown verbatim | CNPG's cron is six-field (seconds first) and is never translated | + +Problems come from Radar's Issues engine (the same detections as `/issues`) plus the audit's `cnpgNoDeclarativeBackup`, worded "No declarative backup schedule" because that is all it proves. A cluster **needs attention** when it has an issue of warning or worse on itself, an instance Pod, or an object that references it. + +## Access + +All data comes from `GET /api/cnpg/workspace`, authorized **per kind**: namespaced kinds use a cluster-wide `list` or fall back per namespace; `ClusterImageCatalog` needs a cluster-scope `list`. Each kind reports coverage (`full`, `partial` with the namespaces read, `denied`, `syncing`, `error`, `notInstalled`). Issues and audit findings are withheld where the underlying kind is not covered — Pod evidence only reaches callers who can list Pods. Denied namespaces are named only when the caller supplied the namespace list. A partial or denied kind makes the screen show a coverage notice, and its facts read "No access" rather than none. + +`GET /api/cnpg/operator` reads operator and plugin Deployments (label `app.kubernetes.io/name=cloudnative-pg`, plugin Services labelled `cnpg.io/pluginName`) and the operator's config references. It ignores the namespace view filter (the operator lives in its own namespace), returns ConfigMap data only with `get configmaps`, and never reads Secrets. + +Cluster logs (`/api/cnpg/clusters/{ns}/{name}/logs`) need `get pods/log`; Activity (`.../activity`) drops events for kinds the caller cannot list. Deleted child objects stay attributed to their Cluster because Radar records the owning cluster on timeline events at ingestion; history recorded before that is marked incomplete. + +## Not in this version + +Runtime data (replication lag, sessions, locks, WAL and slots via the instance manager or Prometheus), Pooler pressure, and operations (Backup now, Switchover, Restart, Hibernate, Restore). See `docs/plans/CNPG_WORKSPACE.md`. + +## Testing + +`make cnpg-demo` (read `scripts/cnpg-demo/README.md` first) produces WAL archiving failure, failed and unrecognised-phase Backups, failing declarations, a Pooler, both catalog kinds and an ObjectStore with a failing server — every state the workspace distinguishes, except successful restores. diff --git a/docs/integrations.md b/docs/integrations.md index 055fb7f94..8e05379fc 100644 --- a/docs/integrations.md +++ b/docs/integrations.md @@ -851,6 +851,8 @@ The source contract is Strimzi's [KafkaConnector status schema](https://strimzi. [CloudNativePG](https://cloudnative-pg.io/) (CNPG) is the Kubernetes operator for PostgreSQL, covering the full lifecycle from bootstrapping to monitoring, with high availability, automated failover, and backup management. +Beyond the per-kind views below, the CloudNativePG **workspace** (`/cnpg`) composes them into fleet, protection, declaration, pooling and operator screens — see [cnpg.md](cnpg.md). + ### What Radar Shows **Cluster Detail View:** diff --git a/internal/server/cnpg_cluster_activity.go b/internal/server/cnpg_cluster_activity.go new file mode 100644 index 000000000..b8535287a --- /dev/null +++ b/internal/server/cnpg_cluster_activity.go @@ -0,0 +1,228 @@ +package server + +import ( + "net/http" + "sort" + "strconv" + "strings" + "time" + + "github.com/go-chi/chi/v5" + + "github.com/skyhook-io/radar/internal/k8s" + "github.com/skyhook-io/radar/internal/timeline" + "github.com/skyhook-io/radar/pkg/resourceid" + pkgtimeline "github.com/skyhook-io/radar/pkg/timeline" +) + +const ( + cnpgActivityDefaultWindow = 24 * time.Hour + cnpgActivityDefaultLimit = 200 + cnpgActivityMaxLimit = 1000 + // cnpgActivityScanLimit bounds the rows read to attribute history. A + // namespace that outgrows it reports truncated rather than silently + // dropping its oldest attribution. + cnpgActivityScanLimit = 10000 +) + +// CNPGClusterActivityResponse is GET /api/cnpg/clusters/{namespace}/{name}/activity. +// Oldest is the earliest row the store still holds for the namespace — the +// floor below which absence means "not retained", not "didn't happen". +// AttributionSince is the earliest visible row that carries this Cluster's +// retained cnpg.io/cluster attribution; before it, deleted children cannot be +// attributed. Both are null when nothing is held. +type CNPGClusterActivityResponse struct { + Events []timeline.TimelineEvent `json:"events"` + Oldest *time.Time `json:"oldest"` + AttributionSince *time.Time `json:"attributionSince"` + Truncated bool `json:"truncated"` +} + +type cnpgActivityKind struct { + group, resource string +} + +// cnpgActivityKinds are the kinds whose rows can belong to one Cluster: the +// Cluster itself, its instance Pods, and every namespaced CNPG kind. +var cnpgActivityKinds = func() map[string]cnpgActivityKind { + out := map[string]cnpgActivityKind{"/Pod": {group: "", resource: "pods"}} + for _, k := range cnpgWorkspaceKinds { + if !k.clusterScoped { + out[k.group+"/"+k.kind] = cnpgActivityKind{group: k.group, resource: k.resource} + } + } + return out +}() + +func cnpgActivityKindNames() []string { + seen := map[string]bool{} + var out []string + for key := range cnpgActivityKinds { + _, kind, _ := strings.Cut(key, "/") + if !seen[kind] { + seen[kind] = true + out = append(out, kind) + } + } + sort.Strings(out) + return out +} + +// cnpgRowAttribution decides whether a timeline row is about the named +// Cluster. Rows about the Cluster match by identity; instance Pods by their +// controller owner; CNPG children by the retained cnpg.io/cluster label, which +// survives their deletion. +func cnpgRowAttribution(e *timeline.TimelineEvent, name string) (matched, labelled bool) { + group := resourceid.GroupFromAPIVersion(e.APIVersion) + if _, ok := cnpgActivityKinds[group+"/"+e.Kind]; !ok { + return false, false + } + labelled = e.Labels[pkgtimeline.CNPGClusterLabel] == name + switch { + case e.Kind == "Cluster" && group == cnpgGroup: + return e.Name == name, false + case e.Kind == "Pod" && group == "": + o := e.Owner + owned := o != nil && o.Kind == "Cluster" && o.Name == name && resourceid.GroupFromAPIVersion(o.APIVersion) == cnpgGroup + return owned, owned && labelled + default: + return labelled, labelled + } +} + +// handleCNPGClusterActivity serves the Cluster's history from the timeline +// store: the Cluster, its instance Pods, and the CNPG objects attributed to it +// — including ones since deleted — with the K8s Events about each. Rows about +// a kind the caller cannot list in the namespace are dropped. +func (s *Server) handleCNPGClusterActivity(w http.ResponseWriter, r *http.Request) { + namespace, name := chi.URLParam(r, "namespace"), chi.URLParam(r, "name") + if !s.requireConnected(w) { + return + } + if noNamespaceAccess(s.getUserNamespaces(r, []string{namespace})) { + s.writeError(w, http.StatusForbidden, "no access to namespace "+namespace) + return + } + if !s.canRead(r, cnpgGroup, "clusters", namespace, "get") { + s.writeError(w, http.StatusForbidden, "no access to clusters.postgresql.cnpg.io in namespace "+namespace) + return + } + + now := time.Now() + since := now.Add(-cnpgActivityDefaultWindow) + if raw := r.URL.Query().Get("since"); raw != "" { + t, err := time.Parse(time.RFC3339, raw) + if err != nil { + s.writeError(w, http.StatusBadRequest, "invalid since "+strconv.Quote(raw)+" (expected RFC3339)") + return + } + since = t + } + limit := cnpgActivityDefaultLimit + if raw := r.URL.Query().Get("limit"); raw != "" { + n, err := strconv.Atoi(raw) + if err != nil || n <= 0 { + s.writeError(w, http.StatusBadRequest, "invalid limit "+strconv.Quote(raw)+" (expected a positive integer)") + return + } + limit = min(n, cnpgActivityMaxLimit) + } + + store := timeline.GetStore() + if store == nil { + s.writeError(w, http.StatusServiceUnavailable, "Timeline store not available") + return + } + clusterContext := k8s.ActiveClusterContext() + rows, err := store.Query(r.Context(), timeline.QueryOptions{ + Namespaces: []string{namespace}, + Kinds: cnpgActivityKindNames(), + APIGroups: []string{"", cnpgGroup, cnpgBarmanGroup}, + ClusterContext: clusterContext, + IncludeManaged: true, + IncludeK8sEvents: true, + Limit: cnpgActivityScanLimit, + }) + if err != nil { + s.writeError(w, http.StatusInternalServerError, err.Error()) + return + } + scanCapped := len(rows) >= cnpgActivityScanLimit + + // Attribution is carried by the subject's own rows; K8s Event rows about + // a subject whose enrichment was already gone carry only its UID. + attributedUIDs := map[string]bool{} + matched := make([]bool, len(rows)) + labelled := make([]bool, len(rows)) + for i := range rows { + matched[i], labelled[i] = cnpgRowAttribution(&rows[i], name) + if matched[i] && rows[i].UID != "" { + attributedUIDs[rows[i].UID] = true + } + } + + allowed := map[string]bool{} + canList := func(e *timeline.TimelineEvent) bool { + key := resourceid.GroupFromAPIVersion(e.APIVersion) + "/" + e.Kind + ok, seen := allowed[key] + if !seen { + target, known := cnpgActivityKinds[key] + ok = known && s.canRead(r, target.group, target.resource, namespace, "list") + allowed[key] = ok + } + return ok + } + + resp := CNPGClusterActivityResponse{Events: []timeline.TimelineEvent{}} + seenIDs := map[string]bool{} + var windowed []timeline.TimelineEvent + for i := range rows { + e := &rows[i] + if !matched[i] && (e.UID == "" || !attributedUIDs[e.UID]) { + continue + } + if !canList(e) || seenIDs[e.ID] { + continue + } + seenIDs[e.ID] = true + if labelled[i] && (resp.AttributionSince == nil || e.Timestamp.Before(*resp.AttributionSince)) { + t := e.Timestamp.UTC() + resp.AttributionSince = &t + } + if e.Timestamp.Before(since) { + continue + } + windowed = append(windowed, *e) + } + sort.SliceStable(windowed, func(i, j int) bool { + if !windowed[i].Timestamp.Equal(windowed[j].Timestamp) { + return windowed[i].Timestamp.After(windowed[j].Timestamp) + } + return windowed[i].ID < windowed[j].ID + }) + resp.Truncated = scanCapped || len(windowed) > limit + if len(windowed) > limit { + windowed = windowed[:limit] + } + if windowed != nil { + resp.Events = windowed + } + + oldest, err := store.Query(r.Context(), timeline.QueryOptions{ + Namespaces: []string{namespace}, + ClusterContext: clusterContext, + IncludeManaged: true, + IncludeK8sEvents: true, + SequenceOrder: timeline.SequenceOrderAscending, + Limit: 1, + }) + if err != nil { + s.writeError(w, http.StatusInternalServerError, err.Error()) + return + } + if len(oldest) > 0 { + t := oldest[0].Timestamp.UTC() + resp.Oldest = &t + } + s.writeJSON(w, resp) +} diff --git a/internal/server/cnpg_cluster_history_test.go b/internal/server/cnpg_cluster_history_test.go new file mode 100644 index 000000000..d781f850e --- /dev/null +++ b/internal/server/cnpg_cluster_history_test.go @@ -0,0 +1,431 @@ +package server + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/client-go/kubernetes" + "k8s.io/client-go/rest" + + "github.com/skyhook-io/radar/internal/auth" + "github.com/skyhook-io/radar/internal/k8s" + "github.com/skyhook-io/radar/internal/timeline" + pkgtimeline "github.com/skyhook-io/radar/pkg/timeline" +) + +func seedCNPGLogCluster(t *testing.T, ns string) { + t.Helper() + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, + withUID(cnpgObj("postgresql.cnpg.io/v1", "Cluster", ns, "pg-orders", map[string]any{"instances": int64(2)}, nil), "orders-uid"), + cnpgObj("cluster.x-k8s.io/v1beta1", "Cluster", ns, "capi-only", nil, nil), + ) + owner := metav1.OwnerReference{APIVersion: "postgresql.cnpg.io/v1", Kind: "Cluster", Name: "pg-orders", UID: "orders-uid", Controller: boolPtr(true)} + stale := owner + stale.UID = "previous-incarnation" + replica := cnpgPod(ns, "pg-orders-2", "pg-orders", owner) + replica.Labels["cnpg.io/instanceRole"] = "replica" + seedCNPGPods(t, + cnpgPod(ns, "pg-orders-1", "pg-orders", owner), + replica, + cnpgPod(ns, "pg-orders-impostor", "pg-orders"), + cnpgPod(ns, "pg-orders-orphan", "pg-orders", stale), + ) +} + +func getCNPGLogs(t *testing.T, path string) (int, CNPGClusterLogsResponse, string) { + t.Helper() + resp, err := http.Get(testServer.URL + path) + if err != nil { + t.Fatalf("GET %s: %v", path, err) + } + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + var out CNPGClusterLogsResponse + if resp.StatusCode == http.StatusOK { + if err := json.Unmarshal(body, &out); err != nil { + t.Fatalf("decode: %v (%s)", err, body) + } + } + return resp.StatusCode, out, string(body) +} + +// useLogServer points Radar's client at an apiserver that serves one JSON log +// line per Pod, so the handler's merge and parse run end to end. +func useLogServer(t *testing.T) { + t.Helper() + apiserver := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + parts := strings.Split(r.URL.Path, "/") + if len(parts) < 2 || parts[len(parts)-1] != "log" { + http.NotFound(w, r) + return + } + pod := parts[len(parts)-2] + fmt.Fprintf(w, "2026-09-28T14:19:58.5Z {\"level\":\"info\",\"logger\":\"postgres\",\"msg\":\"record\",\"record\":{\"error_severity\":\"LOG\",\"message\":\"hello from %s\"}}\n", pod) + })) + t.Cleanup(apiserver.Close) + client, err := kubernetes.NewForConfig(&rest.Config{Host: apiserver.URL}) + if err != nil { + t.Fatal(err) + } + previous := k8s.SetTestClient(client) + t.Cleanup(func() { k8s.SetTestClient(previous) }) +} + +func TestCNPGClusterLogs_OnlyValidatedInstancesContribute(t *testing.T) { + seedCNPGLogCluster(t, "pglogs") + useLogServer(t) + + status, got, body := getCNPGLogs(t, "/api/cnpg/clusters/pglogs/pg-orders/logs") + if status != http.StatusOK { + t.Fatalf("status = %d: %s", status, body) + } + if got.UID != "orders-uid" || got.CapturedAt == "" || got.EmptyMessage == "" { + t.Fatalf("envelope = %+v", got) + } + var names []string + for _, p := range got.Pods { + names = append(names, p.Name) + } + if strings.Join(names, ",") != "pg-orders-1,pg-orders-2" { + t.Fatalf("pods = %v, want only the owned instances", names) + } + for _, entry := range got.Logs { + if entry.Pod != "pg-orders-1" && entry.Pod != "pg-orders-2" { + t.Errorf("log from a non-instance Pod: %+v", entry) + } + if entry.Container != "postgres" { + t.Errorf("container = %q, want the postgres default", entry.Container) + } + } + if len(got.Logs) != 2 { + t.Fatalf("logs = %+v, want one line per instance", got.Logs) + } + for _, entry := range got.Logs { + if entry.Level != "LOG" || entry.Logger != "postgres" || entry.Message != "hello from "+entry.Pod || !strings.HasPrefix(entry.Content, "{") { + t.Errorf("parsed entry = %+v", entry) + } + } + if got.SourceLabels["pg-orders-1"] != "primary" || got.SourceLabels["pg-orders-2"] != "replica" { + t.Errorf("sourceLabels = %v", got.SourceLabels) + } + + status, got, body = getCNPGLogs(t, "/api/cnpg/clusters/pglogs/pg-orders/logs?pod=pg-orders-2") + if status != http.StatusOK || len(got.Pods) != 1 || got.Pods[0].Name != "pg-orders-2" { + t.Fatalf("pod filter: status=%d pods=%+v body=%s", status, got.Pods, body) + } + for _, bad := range []string{"pg-orders-impostor", "pg-orders-orphan", "nope"} { + if status, _, _ := getCNPGLogs(t, "/api/cnpg/clusters/pglogs/pg-orders/logs?pod="+bad); status != http.StatusBadRequest { + t.Errorf("pod=%s: status = %d, want 400", bad, status) + } + } + if status, _, _ := getCNPGLogs(t, "/api/cnpg/clusters/pglogs/pg-orders/logs?sinceTime=yesterday"); status != http.StatusBadRequest { + t.Errorf("bad sinceTime: status = %d, want 400", status) + } +} + +func TestCNPGClusterLogs_NotFound(t *testing.T) { + seedCNPGLogCluster(t, "pglogs404") + for _, path := range []string{ + "/api/cnpg/clusters/pglogs404/missing/logs", + "/api/cnpg/clusters/pglogs404/capi-only/logs", + "/api/cnpg/clusters/elsewhere/pg-orders/logs", + "/api/cnpg/clusters/pglogs404/missing/logs/stream", + } { + if status, _, body := getCNPGLogs(t, path); status != http.StatusNotFound { + t.Errorf("%s: status = %d, want 404 (%s)", path, status, body) + } + } +} + +func TestCNPGClusterLogs_Authorization(t *testing.T) { + seedCNPGLogCluster(t, "pglogsauth") + env := newAuthTestServer(t) + for _, u := range []struct { + name string + clusters bool + }{{"no-clusters", false}, {"no-logs", true}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pglogsauth"}} + perms.SetCanI("get", cnpgGroup, "clusters", "pglogsauth", u.clusters) + allow(perms, "", "pods", "pglogsauth", true) + env.srv.permCache.Set(u.name, nil, perms) + } + for _, tc := range []struct{ user, path, want string }{ + {"no-clusters", "/api/cnpg/clusters/pglogsauth/pg-orders/logs", "clusters.postgresql.cnpg.io"}, + {"no-clusters", "/api/cnpg/clusters/pglogsauth/missing/logs", "clusters.postgresql.cnpg.io"}, + {"no-logs", "/api/cnpg/clusters/pglogsauth/pg-orders/logs", "get pods/log"}, + {"no-logs", "/api/cnpg/clusters/pglogsauth/pg-orders/logs/stream", "get pods/log"}, + } { + resp := env.authGet(t, tc.path, tc.user, "") + body, _ := io.ReadAll(resp.Body) + resp.Body.Close() + if resp.StatusCode != http.StatusForbidden || !strings.Contains(string(body), tc.want) { + t.Errorf("%s %s: status=%d body=%s, want 403 naming %q", tc.user, tc.path, resp.StatusCode, body, tc.want) + } + } +} + +func TestAnnotateCNPGLogEntry(t *testing.T) { + cases := []struct { + name, content, level, logger, message string + }{ + { + name: "postgres record", + content: `{"level":"info","ts":"2026-09-28T14:19:58.123Z","logger":"postgres","msg":"record","record":{"error_severity":"FATAL","message":"password authentication failed","log_time":"2026-09-28 14:19:58.123 UTC"}}`, + level: "FATAL", logger: "postgres", message: "password authentication failed", + }, + { + name: "instance manager error", + content: `{"level":"error","ts":"2026-09-28T14:19:58Z","logger":"barman-cloud-wal-archive","msg":"Error invoking barman-cloud-wal-archive","error":"exit status 4"}`, + level: "ERROR", logger: "barman-cloud-wal-archive", message: "Error invoking barman-cloud-wal-archive: exit status 4", + }, + { + name: "structured error", + content: `{"level":"error","msg":"failed","error":{"code":2}}`, + level: "ERROR", message: `failed: {"code":2}`, + }, + {name: "plain text", content: "LOG: database system is ready"}, + {name: "broken json", content: `{"level":"info"`}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + entry := workloadLogEntry{Content: tc.content} + annotateCNPGLogEntry(&entry) + if entry.Level != tc.level || entry.Logger != tc.logger || entry.Message != tc.message || entry.Content != tc.content { + t.Fatalf("got level=%q logger=%q message=%q content-changed=%v", entry.Level, entry.Logger, entry.Message, entry.Content != tc.content) + } + }) + } + raw, _ := json.Marshal(workloadLogEntry{Pod: "p", Content: "x"}) + if strings.Contains(string(raw), "level") || strings.Contains(string(raw), "message") { + t.Fatalf("unparsed entries grew fields: %s", raw) + } +} + +func TestParseCNPGLogQuery(t *testing.T) { + now := time.Date(2026, 9, 28, 12, 0, 0, 0, time.UTC) + req := httptest.NewRequest("GET", "/?sinceTime=2026-09-28T11:59:00.5Z", nil) + if _, err := parseCNPGLogQuery(req, now); err != nil { + t.Fatalf("fractional RFC3339 rejected: %v", err) + } + req = httptest.NewRequest("GET", "/?sinceTime=2026-09-28T11:58:30Z", nil) + q, err := parseCNPGLogQuery(req, now) + if err != nil || q.sinceSeconds == nil || *q.sinceSeconds != 90 || q.container != "postgres" || q.tailLines != 200 { + t.Fatalf("query = %+v err=%v", q, err) + } + if q.keep(workloadLogEntry{Timestamp: "2026-09-28T11:58:29.9Z"}) || !q.keep(workloadLogEntry{Timestamp: "2026-09-28T11:58:30Z"}) { + t.Fatal("sinceTime overlap not trimmed") + } + req = httptest.NewRequest("GET", "/?sinceTime=2026-09-28T11:58:30Z&sinceSeconds=5", nil) + if _, err := parseCNPGLogQuery(req, now); err == nil { + t.Fatal("sinceTime with sinceSeconds accepted") + } +} + +func useMemoryTimeline(t *testing.T) timeline.EventStore { + t.Helper() + timeline.ResetStore() + if err := timeline.InitStore(timeline.StoreConfig{Type: timeline.StoreTypeMemory, MaxSize: 1000}); err != nil { + t.Fatalf("InitStore: %v", err) + } + t.Cleanup(func() { + timeline.ResetStore() + if err := timeline.InitStore(timeline.DefaultStoreConfig()); err != nil { + t.Fatalf("re-init global store: %v", err) + } + }) + return timeline.GetStore() +} + +type activityRow struct { + id, apiVersion, kind, name, uid string + source timeline.EventSource + eventType timeline.EventType + age time.Duration + labels map[string]string + owner *timeline.OwnerInfo +} + +func seedActivity(t *testing.T, store timeline.EventStore, ns string, rows ...activityRow) { + t.Helper() + now := time.Now() + for _, r := range rows { + source, eventType := r.source, r.eventType + if source == "" { + source = timeline.SourceInformer + } + if eventType == "" { + eventType = timeline.EventTypeUpdate + } + e := timeline.TimelineEvent{ + ID: r.id, Timestamp: now.Add(-r.age), Source: source, Kind: r.kind, APIVersion: r.apiVersion, + Namespace: ns, Name: r.name, UID: r.uid, EventType: eventType, Labels: r.labels, Owner: r.owner, + ClusterContext: k8s.ActiveClusterContext(), + } + if err := store.Append(context.Background(), e); err != nil { + t.Fatalf("append %s: %v", r.id, err) + } + } +} + +func cnpgActivityFixture(t *testing.T, ns string) { + t.Helper() + store := useMemoryTimeline(t) + attributed := map[string]string{pkgtimeline.CNPGClusterLabel: "pg-orders"} + clusterOwner := &timeline.OwnerInfo{Kind: "Cluster", Name: "pg-orders", APIVersion: "postgresql.cnpg.io/v1", UID: "orders-uid"} + seedActivity(t, store, ns, + activityRow{id: "cluster-update", apiVersion: "postgresql.cnpg.io/v1", kind: "Cluster", name: "pg-orders", uid: "orders-uid", age: time.Hour}, + activityRow{id: "backup-add", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-orders-b1", uid: "b1", eventType: timeline.EventTypeAdd, age: 50 * time.Minute, labels: attributed}, + activityRow{id: "backup-delete", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-orders-b1", uid: "b1", eventType: timeline.EventTypeDelete, age: 40 * time.Minute, labels: attributed}, + activityRow{id: "backup-k8s-event", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-orders-b1", uid: "b1", source: timeline.SourceK8sEvent, eventType: timeline.EventTypeWarning, age: 45 * time.Minute}, + activityRow{id: "pod-update", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p1", age: 30 * time.Minute, labels: attributed, owner: clusterOwner}, + activityRow{id: "pod-k8s-event", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p1", source: timeline.SourceK8sEvent, eventType: timeline.EventTypeWarning, age: 20 * time.Minute}, + activityRow{id: "old-pooler", apiVersion: "postgresql.cnpg.io/v1", kind: "Pooler", name: "pg-orders-rw", uid: "pool1", age: 48 * time.Hour, labels: attributed}, + activityRow{id: "other-backup", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-other-b1", uid: "b2", age: 10 * time.Minute, labels: map[string]string{pkgtimeline.CNPGClusterLabel: "pg-other"}}, + activityRow{id: "velero-backup", apiVersion: "velero.io/v1", kind: "Backup", name: "pg-orders", uid: "v1", age: 10 * time.Minute, labels: attributed}, + activityRow{id: "capi-cluster", apiVersion: "cluster.x-k8s.io/v1beta1", kind: "Cluster", name: "pg-orders", uid: "capi", age: 10 * time.Minute}, + activityRow{id: "impostor-pod", apiVersion: "v1", kind: "Pod", name: "impostor", uid: "p9", age: 10 * time.Minute, labels: attributed}, + ) + seedActivity(t, store, "elsewhere", + activityRow{id: "elsewhere-backup", apiVersion: "postgresql.cnpg.io/v1", kind: "Backup", name: "pg-orders-b9", uid: "b9", age: 10 * time.Minute, labels: attributed}, + ) +} + +func decodeActivity(t *testing.T, resp *http.Response) CNPGClusterActivityResponse { + t.Helper() + defer resp.Body.Close() + body, _ := io.ReadAll(resp.Body) + if resp.StatusCode != http.StatusOK { + t.Fatalf("status = %d: %s", resp.StatusCode, body) + } + var out CNPGClusterActivityResponse + if err := json.Unmarshal(body, &out); err != nil { + t.Fatalf("decode: %v", err) + } + return out +} + +func activityIDs(resp CNPGClusterActivityResponse) []string { + var out []string + for _, e := range resp.Events { + out = append(out, e.ID) + } + return out +} + +func TestCNPGClusterActivity_AttributesDeletedChildren(t *testing.T) { + cnpgActivityFixture(t, "pgact") + resp, err := http.Get(testServer.URL + "/api/cnpg/clusters/pgact/pg-orders/activity") + if err != nil { + t.Fatal(err) + } + got := decodeActivity(t, resp) + want := "pod-k8s-event,pod-update,backup-delete,backup-k8s-event,backup-add,cluster-update" + if strings.Join(activityIDs(got), ",") != want { + t.Fatalf("events = %v, want %s", activityIDs(got), want) + } + if got.Truncated { + t.Error("truncated on a small history") + } + if got.Oldest == nil || got.AttributionSince == nil { + t.Fatalf("oldest=%v attributionSince=%v", got.Oldest, got.AttributionSince) + } + if age := time.Since(*got.AttributionSince); age < 47*time.Hour { + t.Errorf("attributionSince = %v, want the 48h-old Pooler row outside the window", got.AttributionSince) + } + + resp, _ = http.Get(testServer.URL + "/api/cnpg/clusters/pgact/pg-orders/activity?limit=2&since=" + time.Now().Add(-72*time.Hour).UTC().Format(time.RFC3339)) + got = decodeActivity(t, resp) + if len(got.Events) != 2 || !got.Truncated || got.Events[0].ID != "pod-k8s-event" { + t.Fatalf("limited: events=%v truncated=%v", activityIDs(got), got.Truncated) + } + + for _, q := range []string{"?since=yesterday", "?limit=0", "?limit=x"} { + resp, _ := http.Get(testServer.URL + "/api/cnpg/clusters/pgact/pg-orders/activity" + q) + resp.Body.Close() + if resp.StatusCode != http.StatusBadRequest { + t.Errorf("%s: status = %d, want 400", q, resp.StatusCode) + } + } +} + +func TestCNPGClusterActivity_DropsKindsTheCallerCannotList(t *testing.T) { + cnpgActivityFixture(t, "pgactauth") + env := newAuthTestServer(t) + for _, u := range []struct { + name string + clusterGet bool + backupsListed bool + }{{"reader", true, true}, {"no-backups", true, false}, {"no-clusters", false, true}} { + perms := &auth.UserPermissions{AllowedNamespaces: []string{"pgactauth"}} + perms.SetCanI("get", cnpgGroup, "clusters", "pgactauth", u.clusterGet) + allow(perms, cnpgGroup, "clusters", "pgactauth", true) + allow(perms, cnpgGroup, "backups", "pgactauth", u.backupsListed) + allow(perms, cnpgGroup, "poolers", "pgactauth", true) + allow(perms, "", "pods", "pgactauth", true) + env.srv.permCache.Set(u.name, nil, perms) + } + + control := decodeActivity(t, env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "reader", "")) + if !strings.Contains(strings.Join(activityIDs(control), ","), "backup-delete") { + t.Fatalf("control: deleted Backup missing: %v", activityIDs(control)) + } + + got := decodeActivity(t, env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "no-backups", "")) + for _, e := range got.Events { + if e.Kind == "Backup" { + t.Errorf("Backup row reached a caller who cannot list backups: %s", e.ID) + } + } + if len(got.Events) != 3 { + t.Errorf("events = %v, want the Cluster and Pod rows", activityIDs(got)) + } + + resp := env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "no-clusters", "") + resp.Body.Close() + if resp.StatusCode != http.StatusForbidden { + t.Errorf("no cluster get: status = %d, want 403", resp.StatusCode) + } +} + +func TestCNPGClusterLogsStream_SendsParsedInstanceLines(t *testing.T) { + seedCNPGLogCluster(t, "pgstream") + useLogServer(t) + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + req, _ := http.NewRequestWithContext(ctx, "GET", testServer.URL+"/api/cnpg/clusters/pgstream/pg-orders/logs/stream?pod=pg-orders-2", nil) + resp, err := http.DefaultClient.Do(req) + if err != nil { + t.Fatal(err) + } + defer resp.Body.Close() + if resp.StatusCode != http.StatusOK || resp.Header.Get("Content-Type") != "text/event-stream" { + t.Fatalf("status=%d content-type=%q", resp.StatusCode, resp.Header.Get("Content-Type")) + } + buf := make([]byte, 0, 4096) + chunk := make([]byte, 1024) + for !strings.Contains(string(buf), "event: log") { + n, err := resp.Body.Read(chunk) + buf = append(buf, chunk[:n]...) + if err != nil { + t.Fatalf("stream ended before a log event: %v\n%s", err, buf) + } + } + stream := string(buf) + sawConnected := strings.Contains(stream, "event: connected") && strings.Contains(stream, `"name":"pg-orders-2"`) + if !sawConnected || strings.Contains(stream, `"name":"pg-orders-1"`) { + t.Fatalf("connected event wrong:\n%s", stream) + } + for _, want := range []string{`"level":"LOG"`, `"message":"hello from pg-orders-2"`, `"sourceLabel":"replica"`} { + if !strings.Contains(stream, want) { + t.Errorf("stream missing %s:\n%s", want, stream) + } + } +} diff --git a/internal/server/cnpg_cluster_logs.go b/internal/server/cnpg_cluster_logs.go new file mode 100644 index 000000000..274a70b3a --- /dev/null +++ b/internal/server/cnpg_cluster_logs.go @@ -0,0 +1,465 @@ +package server + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "log" + "math" + "net/http" + "sort" + "strings" + "sync" + "time" + + "github.com/go-chi/chi/v5" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/labels" + "k8s.io/apimachinery/pkg/types" + + "github.com/skyhook-io/radar/internal/k8s" +) + +const ( + cnpgClusterLabel = "cnpg.io/cluster" + cnpgDefaultLogContainer = "postgres" + cnpgDefaultLogTailLines = 200 + cnpgLogDiscoveryInterval = 5 * time.Second + cnpgLogsEmptyMessage = "No readable logs from this cluster's instances in this snapshot. Refresh after the instances start." + cnpgLogsNoInstanceMessage = "This cluster has no instance Pods yet." +) + +// CNPGClusterLogsResponse is GET /api/cnpg/clusters/{namespace}/{name}/logs. +// Pods and SourceLabels list only the instances that contributed a source to +// this snapshot; SourceLabels maps a Pod to its instance role. +type CNPGClusterLogsResponse struct { + UID types.UID `json:"uid"` + Pods []WorkloadPodInfo `json:"pods"` + Logs []workloadLogEntry `json:"logs"` + Notice string `json:"notice"` + SourceLabels map[string]string `json:"sourceLabels,omitempty"` + CapturedAt string `json:"capturedAt"` + EmptyMessage string `json:"emptyMessage"` +} + +type cnpgLogQuery struct { + container string + tailLines int64 + sinceSeconds *int64 + sinceTime time.Time + pod string +} + +func parseCNPGLogQuery(r *http.Request, now time.Time) (cnpgLogQuery, error) { + q := r.URL.Query() + out := cnpgLogQuery{ + container: q.Get("container"), + tailLines: parseTailLines(q.Get("tailLines"), cnpgDefaultLogTailLines), + sinceSeconds: parseSinceSeconds(q.Get("sinceSeconds")), + pod: q.Get("pod"), + } + if out.container == "" { + out.container = cnpgDefaultLogContainer + } + if raw := q.Get("sinceTime"); raw != "" { + if q.Get("sinceSeconds") != "" { + return out, errors.New("sinceSeconds and sinceTime are mutually exclusive") + } + t, err := time.Parse(time.RFC3339, raw) + if err != nil { + return out, fmt.Errorf("invalid sinceTime %q (expected RFC3339)", raw) + } + out.sinceTime = t + // The pod log API takes whole seconds; round up and trim the overlap + // from the entries afterwards. + secs := max(int64(math.Ceil(now.Sub(t).Seconds())), 1) + out.sinceSeconds = &secs + } + return out, nil +} + +func (q cnpgLogQuery) keep(entry workloadLogEntry) bool { + if q.sinceTime.IsZero() || entry.Timestamp == "" { + return true + } + ts, err := time.Parse(time.RFC3339Nano, entry.Timestamp) + return err != nil || !ts.Before(q.sinceTime) +} + +// authorizeCNPGClusterLogs gates on reading the Cluster, listing its Pods and +// reading their logs — before the Cluster is looked up, so a denied caller +// cannot probe which Clusters exist. +func (s *Server) authorizeCNPGClusterLogs(w http.ResponseWriter, r *http.Request, namespace string) bool { + if !s.requireConnected(w) { + return false + } + if noNamespaceAccess(s.getUserNamespaces(r, []string{namespace})) { + s.writeError(w, http.StatusForbidden, "no access to namespace "+namespace) + return false + } + if !s.canRead(r, cnpgGroup, "clusters", namespace, "get") { + s.writeError(w, http.StatusForbidden, "no access to clusters.postgresql.cnpg.io in namespace "+namespace) + return false + } + if !s.canRead(r, "", "pods", namespace, "list") { + s.writeError(w, http.StatusForbidden, "no access to pods in namespace "+namespace) + return false + } + return s.authorizePodLogRead(w, r, namespace) +} + +// loadCNPGCluster reads one CNPG Cluster from the dynamic cache. The error is +// already written when ok is false. +func (s *Server) loadCNPGCluster(w http.ResponseWriter, r *http.Request, cache *k8s.ResourceCache, namespace, name string) (*unstructured.Unstructured, bool) { + cluster, err := findCNPGCluster(r.Context(), cache, namespace, name) + switch { + case err == nil && cluster != nil: + return cluster, true + case err == nil, errors.Is(err, k8s.ErrUnknownDynamicKind): + s.writeError(w, http.StatusNotFound, "CloudNativePG Cluster "+namespace+"/"+name+" not found") + case errors.Is(err, errDynamicNotSynced): + s.writeError(w, http.StatusServiceUnavailable, "CloudNativePG Clusters are still syncing") + default: + log.Printf("[cnpg] Failed to read Cluster %s/%s: %v", namespace, name, err) + s.writeError(w, http.StatusInternalServerError, "failed to read CloudNativePG Cluster") + } + return nil, false +} + +func findCNPGCluster(ctx context.Context, cache *k8s.ResourceCache, namespace, name string) (*unstructured.Unstructured, error) { + clusters, err := filterCNPGGroup(listDynamicSynced(ctx, cache, "Cluster", cnpgGroup, namespace)) + if err != nil { + return nil, err + } + for _, c := range clusters { + if c.GetNamespace() == namespace && c.GetName() == name && c.GroupVersionKind().Group == cnpgGroup { + return c, nil + } + } + return nil, nil +} + +// cnpgClusterInstancePods returns the Cluster's instance Pods under the same +// label-and-controller-UID rule the workspace uses, sorted by name. +func cnpgClusterInstancePods(cache *k8s.ResourceCache, cluster *unstructured.Unstructured) ([]*corev1.Pod, error) { + lister := cache.Pods() + if lister == nil { + return nil, errors.New("pod cache unavailable") + } + namespace, name := cluster.GetNamespace(), cluster.GetName() + candidates, err := lister.Pods(namespace).List(labels.SelectorFromSet(labels.Set{cnpgClusterLabel: name})) + if err != nil { + return nil, err + } + uids := map[string]types.UID{namespace + "/" + name: cluster.GetUID()} + pods := make([]*corev1.Pod, 0, len(candidates)) + for _, p := range candidates { + if p != nil && isCNPGInstancePod(p, uids) { + pods = append(pods, p) + } + } + sort.Slice(pods, func(i, j int) bool { return pods[i].Name < pods[j].Name }) + return pods, nil +} + +func cnpgInstanceRole(p *corev1.Pod) string { + if role := p.Labels["cnpg.io/instanceRole"]; role != "" { + return role + } + return p.Labels["role"] +} + +// selectCNPGLogPods narrows to the requested instance. ok is false when the +// requested Pod is not one of the Cluster's instances. +func selectCNPGLogPods(pods []*corev1.Pod, want string) ([]*corev1.Pod, bool) { + if want == "" { + return pods, true + } + for _, p := range pods { + if p.Name == want { + return []*corev1.Pod{p}, true + } + } + return nil, false +} + +// cnpgLogRecord is the subset of a CloudNativePG instance-manager JSON log +// line the viewer surfaces. PostgreSQL's own log lines arrive wrapped, with the +// server's severity and message under record. +type cnpgLogRecord struct { + Level string `json:"level"` + Logger string `json:"logger"` + Msg string `json:"msg"` + Error any `json:"error"` + Record *struct { + ErrorSeverity string `json:"error_severity"` + Message string `json:"message"` + } `json:"record"` +} + +// annotateCNPGLogEntry fills the parsed fields of a CloudNativePG log line and +// leaves anything that is not one untouched. +func annotateCNPGLogEntry(entry *workloadLogEntry) { + content := strings.TrimSpace(entry.Content) + if !strings.HasPrefix(content, "{") { + return + } + var rec cnpgLogRecord + if err := json.Unmarshal([]byte(content), &rec); err != nil { + return + } + level := strings.ToUpper(rec.Level) + message := rec.Msg + if rec.Record != nil { + if rec.Record.ErrorSeverity != "" { + level = rec.Record.ErrorSeverity + } + if rec.Record.Message != "" { + message = rec.Record.Message + } + } + if errText := cnpgLogErrorText(rec.Error); errText != "" { + if message == "" { + message = errText + } else { + message += ": " + errText + } + } + entry.Level, entry.Logger, entry.Message = level, rec.Logger, message +} + +func cnpgLogErrorText(v any) string { + switch e := v.(type) { + case nil: + return "" + case string: + return e + default: + b, err := json.Marshal(e) + if err != nil { + return "" + } + return string(b) + } +} + +// handleCNPGClusterLogs serves GET /api/cnpg/clusters/{namespace}/{name}/logs: +// a bounded snapshot of every instance Pod's logs, merged by timestamp. +func (s *Server) handleCNPGClusterLogs(w http.ResponseWriter, r *http.Request) { + namespace, name := chi.URLParam(r, "namespace"), chi.URLParam(r, "name") + if !s.authorizeCNPGClusterLogs(w, r, namespace) { + return + } + query, qerr := parseCNPGLogQuery(r, time.Now()) + if qerr != nil { + s.writeError(w, http.StatusBadRequest, qerr.Error()) + return + } + cache := k8s.GetResourceCache() + if cache == nil { + s.writeError(w, http.StatusServiceUnavailable, "resource cache not available") + return + } + cluster, ok := s.loadCNPGCluster(w, r, cache, namespace, name) + if !ok { + return + } + instances, err := cnpgClusterInstancePods(cache, cluster) + if err != nil { + log.Printf("[cnpg] Failed to list instance Pods for %s/%s: %v", namespace, name, err) + s.writeError(w, http.StatusServiceUnavailable, "instance Pods unavailable: "+err.Error()) + return + } + pods, ok := selectCNPGLogPods(instances, query.pod) + if !ok { + s.writeError(w, http.StatusBadRequest, "pod "+query.pod+" is not an instance of CloudNativePG Cluster "+namespace+"/"+name) + return + } + + resp := CNPGClusterLogsResponse{ + UID: cluster.GetUID(), + Pods: []WorkloadPodInfo{}, + Logs: []workloadLogEntry{}, + CapturedAt: time.Now().UTC().Format(time.RFC3339), + EmptyMessage: cnpgLogsEmptyMessage, + } + if len(pods) == 0 { + resp.EmptyMessage = cnpgLogsNoInstanceMessage + s.writeJSON(w, resp) + return + } + client := s.getClientForRequest(r) + if client == nil { + s.writeError(w, http.StatusServiceUnavailable, "cluster client unavailable") + return + } + + snapshot := collectLogsFromPods(r.Context(), client, namespace, pods, query.container, query.tailLines, query.sinceSeconds, true) + shown := []*corev1.Pod{} + sourceLabels := map[string]string{} + for _, p := range pods { + if !snapshot.SourcePods[p.Name] { + continue + } + shown = append(shown, p) + if role := cnpgInstanceRole(p); role != "" { + sourceLabels[p.Name] = role + } + } + for _, entry := range snapshot.Logs { + if !query.keep(entry) { + continue + } + entry.SourceLabel = sourceLabels[entry.Pod] + annotateCNPGLogEntry(&entry) + resp.Logs = append(resp.Logs, entry) + } + sortLogsByTimestamp(resp.Logs) + resp.Pods = buildPodInfos(shown) + resp.Notice = snapshot.Notice + if len(sourceLabels) > 0 { + resp.SourceLabels = sourceLabels + } + s.writeJSON(w, resp) +} + +// handleCNPGClusterLogsStream serves GET +// /api/cnpg/clusters/{namespace}/{name}/logs/stream: an SSE follow of every +// instance Pod, re-resolving instances as the Cluster fails over or scales. +// Events: connected {cluster, namespace, uid, pods}, log (a log entry with +// the parsed fields), pod_added {pods}, pod_removed {pod, reason}, end +// {reason}, error {error}. +func (s *Server) handleCNPGClusterLogsStream(w http.ResponseWriter, r *http.Request) { + namespace, name := chi.URLParam(r, "namespace"), chi.URLParam(r, "name") + if !s.authorizeCNPGClusterLogs(w, r, namespace) { + return + } + query, qerr := parseCNPGLogQuery(r, time.Now()) + if qerr != nil { + s.writeError(w, http.StatusBadRequest, qerr.Error()) + return + } + cache := k8s.GetResourceCache() + if cache == nil { + s.writeError(w, http.StatusServiceUnavailable, "resource cache not available") + return + } + cluster, ok := s.loadCNPGCluster(w, r, cache, namespace, name) + if !ok { + return + } + instances, err := cnpgClusterInstancePods(cache, cluster) + if err != nil { + log.Printf("[cnpg] Failed to list instance Pods for %s/%s: %v", namespace, name, err) + s.writeError(w, http.StatusServiceUnavailable, "instance Pods unavailable: "+err.Error()) + return + } + pods, ok := selectCNPGLogPods(instances, query.pod) + if !ok { + s.writeError(w, http.StatusBadRequest, "pod "+query.pod+" is not an instance of CloudNativePG Cluster "+namespace+"/"+name) + return + } + client := s.getClientForRequest(r) + if client == nil { + s.writeError(w, http.StatusServiceUnavailable, "cluster client unavailable") + return + } + flusher, ok := w.(http.Flusher) + if !ok { + s.writeError(w, http.StatusInternalServerError, "streaming not supported") + return + } + + w.Header().Set("Content-Type", "text/event-stream") + w.Header().Set("Cache-Control", "no-cache") + w.Header().Set("Connection", "keep-alive") + w.Header().Set("X-Accel-Buffering", "no") + + uid := cluster.GetUID() + sendSSEEvent(w, flusher, "connected", map[string]any{ + "cluster": name, "namespace": namespace, "uid": uid, "pods": buildPodInfos(pods), + }) + + ctx, cancel := context.WithCancel(r.Context()) + defer cancel() + logCh := make(chan workloadLogEntry, 1000) + var active sync.Map + roles := map[string]string{} + start := func(pods []*corev1.Pod) { + for _, pod := range pods { + roles[pod.Name] = cnpgInstanceRole(pod) + for _, c := range k8s.GetContainersForPod(pod, query.container, true) { + key := pod.Name + "/" + c + if _, exists := active.Load(key); exists { + continue + } + streamCtx, streamCancel := context.WithCancel(ctx) + active.Store(key, streamCancel) + go func(podName, containerName, key string) { + defer active.Delete(key) + streamPodLogs(streamCtx, client, namespace, podName, containerName, query.tailLines, query.sinceSeconds, logCh) + }(pod.Name, c, key) + } + } + } + start(pods) + + known := map[string]bool{} + for _, p := range pods { + known[p.Name] = true + } + ticker := time.NewTicker(cnpgLogDiscoveryInterval) + defer ticker.Stop() + for { + select { + case <-ctx.Done(): + return + case entry := <-logCh: + if !query.keep(entry) { + continue + } + entry.SourceLabel = roles[entry.Pod] + annotateCNPGLogEntry(&entry) + sendSSEEvent(w, flusher, "log", entry) + case <-ticker.C: + current, err := findCNPGCluster(ctx, cache, namespace, name) + if err != nil { + continue + } + if current == nil || current.GetUID() != uid { + sendSSEEvent(w, flusher, "end", map[string]string{"reason": "cluster deleted"}) + return + } + all, err := cnpgClusterInstancePods(cache, current) + if err != nil { + continue + } + currentPods, _ := selectCNPGLogPods(all, query.pod) + present := map[string]bool{} + for _, p := range currentPods { + present[p.Name] = true + if !known[p.Name] { + known[p.Name] = true + sendSSEEvent(w, flusher, "pod_added", map[string]any{"pods": []WorkloadPodInfo{buildPodInfo(p, time.Now())}}) + } + } + for podName := range known { + if present[podName] { + continue + } + delete(known, podName) + active.Range(func(key, value any) bool { + if strings.HasPrefix(key.(string), podName+"/") { + value.(context.CancelFunc)() + active.Delete(key) + } + return true + }) + sendSSEEvent(w, flusher, "pod_removed", map[string]string{"pod": podName, "reason": "terminated"}) + } + start(currentPods) + } + } +} diff --git a/internal/server/server.go b/internal/server/server.go index 1cf9f20b0..08338b8fc 100644 --- a/internal/server/server.go +++ b/internal/server/server.go @@ -520,6 +520,7 @@ func (s *Server) setupAppRoutes(r chi.Router) { // over a slow cluster link legitimately takes longer than that. r.Post("/pods/{namespace}/{name}/files/save", s.handlePodFileSave) r.Get("/workloads/{kind}/{namespace}/{name}/logs/stream", s.handleWorkloadLogsStream) + r.Get("/cnpg/clusters/{namespace}/{name}/logs/stream", s.handleCNPGClusterLogsStream) // AI investigation event stream via SSE — long-lived; lives outside the // 60s timeout group. The run keeps going server-side after disconnect. r.Get("/diagnose/runs/{id}/stream", s.handleDiagnoseRunStream) @@ -595,6 +596,8 @@ func (s *Server) setupAppRoutes(r chi.Router) { r.Get("/cnpg/operator", s.handleCNPGOperator) r.Get("/cnpg/imagecatalogs/{namespace}/{name}/clusters", s.handleCNPGCatalogUsers) r.Get("/cnpg/clusterimagecatalogs/{name}/clusters", s.handleCNPGCatalogUsers) + r.Get("/cnpg/clusters/{namespace}/{name}/logs", s.handleCNPGClusterLogs) + r.Get("/cnpg/clusters/{namespace}/{name}/activity", s.handleCNPGClusterActivity) r.Get("/velero/backupstoragelocations/{namespace}/{name}/backups", s.handleVeleroStoredBackups) // POST: creates a DownloadRequest, which is the only supported way to // read the messages behind a run's error and warning counts. diff --git a/internal/server/workload_logs.go b/internal/server/workload_logs.go index d35b22167..8f9709ad5 100644 --- a/internal/server/workload_logs.go +++ b/internal/server/workload_logs.go @@ -68,6 +68,10 @@ type workloadLogEntry struct { Timestamp string `json:"timestamp"` Content string `json:"content"` SourceLabel string `json:"sourceLabel,omitempty"` + // Parsed from a structured line by sources that know their log format. + Level string `json:"level,omitempty"` + Logger string `json:"logger,omitempty"` + Message string `json:"message,omitempty"` } type workloadLogMetadata struct { @@ -363,7 +367,7 @@ func (s *Server) authorizeWorkloadLogRead(w http.ResponseWriter, r *http.Request func (s *Server) authorizePodLogRead(w http.ResponseWriter, r *http.Request, namespace string) bool { if !s.canReadSubresource(r, "", "pods", "log", namespace, "get") { - s.writeError(w, http.StatusForbidden, "no access to pod logs in namespace "+namespace) + s.writeError(w, http.StatusForbidden, "no access to pod logs in namespace "+namespace+": requires get pods/log") return false } return true diff --git a/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx index beb3580cb..0902a147d 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGBackupSummary.tsx @@ -1,4 +1,4 @@ -import { cronToHuman, formatDuration } from '../resources/resource-utils' +import { formatDuration } from '../resources/resource-utils' import { CNPG_BARMAN_OBJECTSTORE_GROUP, CNPG_GROUP, @@ -13,6 +13,7 @@ import { backupDestination, backupsForScheduledBackup, clustersIn, + isBackupFromSchedule, refOf, relationUnavailable, scheduledBackupOf, @@ -75,6 +76,10 @@ export function CNPGBackupSummary({ resource, workspace, onNavigate }: SummaryPr const pod = resource?.status?.instanceID?.podName const error = resource?.status?.error const trigger = scheduledBackupOf(resource) + const liveSchedule = trigger + ? workspaceList(workspace, 'scheduledBackups').find((s) => s?.metadata?.namespace === ns && s?.metadata?.name === trigger) + : undefined + const scheduleReplaced = !!liveSchedule && !isBackupFromSchedule(resource, liveSchedule) const dest = backupDestination(resource, clustersIn(workspace)) const method = methodText(resource) @@ -113,7 +118,14 @@ export function CNPGBackupSummary({ resource, workspace, onNavigate }: SummaryPr {trigger ? ( ScheduledBackup{' '} - + {scheduleReplaced ? ( + <> + {trigger} + · an earlier schedule of that name; the current one is a different object + + ) : ( + + )} ) : ( 'On demand' @@ -124,6 +136,7 @@ export function CNPGBackupSummary({ resource, workspace, onNavigate }: SummaryPr ObjectStore{' '} + {dest.inferred && · from the Cluster's current configuration} ) : dest.type === 'path' ? ( {dest.path} @@ -142,7 +155,6 @@ export function CNPGBackupSummary({ resource, workspace, onNavigate }: SummaryPr export function CNPGScheduledBackupSummary({ resource, workspace, onNavigate }: SummaryProps) { const ns = resource?.metadata?.namespace ?? '' const cron = resource?.spec?.schedule - const human = cron ? cronToHuman(cron) : '' const next = getCNPGScheduledBackupNextSchedule(resource) const runsUnavailable = relationUnavailable(workspace, 'backups', ns, 'Backups') const runs = runsUnavailable ? [] : backupsForScheduledBackup(resource, workspaceList(workspace, 'backups')) @@ -164,7 +176,7 @@ export function CNPGScheduledBackupSummary({ resource, workspace, onNavigate }: {cron ? ( {cron} - {human && human !== cron && · {human}} + · six fields, seconds first ) : ( diff --git a/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx index 395c05ca4..dc667382d 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx @@ -64,7 +64,7 @@ function Reconciled({ resource, extra }: { resource: any; extra?: ReactNode }) { function DeclaredIn({ resource }: { resource: any }) { const src = gitopsSourceOf(resource) - if (!src) return Applied directly (no GitOps owner label) + if (!src) return return ( {src.tool === 'argocd' ? 'Argo CD application' : 'Flux'} {src.namespace ? `${src.namespace}/${src.name}` : src.name} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx index 7b89de5da..c7e05904e 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx @@ -17,7 +17,10 @@ export function CNPGImageCatalogSummary({ const ns = resource?.metadata?.namespace ?? '' const entries = getCNPGImageCatalogEntries(resource) const majors = new Set(entries.map((e) => e.major)) - const unavailable = relationUnavailable(workspace, 'clusters', clusterScoped ? undefined : ns, 'Clusters') + // A ClusterImageCatalog's users can be in any namespace, so partial coverage + // lists what was read; an ImageCatalog's users are all in its own namespace. + const partial = clusterScoped && workspace?.coverage?.clusters?.state === 'partial' + const unavailable = partial ? null : relationUnavailable(workspace, 'clusters', clusterScoped ? undefined : ns, 'Clusters') const users = unavailable ? [] : clustersUsingCatalog(resource, clustersIn(workspace)) return ( @@ -68,7 +71,8 @@ export function CNPGImageCatalogSummary({ ))} )} - {clusterScoped && !unavailable && ( + {!unavailable && partial && Only clusters in namespaces you can read are listed.} + {!unavailable && !partial && clusterScoped && ( Among clusters you can see; clusters in namespaces you cannot read are not listed )} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx index bbda3bb7d..39e1556ea 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx @@ -111,21 +111,23 @@ export function CNPGObjectStoreSummary({ ) } > -
- {w.firstRecoverabilityPoint || w.lastSuccessfulBackupTime ? ( - - {utc(w.firstRecoverabilityPoint)} → {utc(w.lastSuccessfulBackupTime)} - - ) : ( - - )} +
+
+ First recoverability point + {w.firstRecoverabilityPoint ? utc(w.firstRecoverabilityPoint) : } +
+
+ Last successful backup + {w.lastSuccessfulBackupTime ? utc(w.lastSuccessfulBackupTime) : } +
{w.lastFailedBackupTime && ( -
- Last failed backup +
+ Last failed backup + {utc(w.lastFailedBackupTime)}
)} {w.failingSinceLastSuccess && ( - The window is still restorable up to the last successful backup; it stops advancing while uploads fail. + {w.lastSuccessfulBackupTime ? 'A backup failed after the last recorded success.' : 'No successful backup recorded.'} )}
diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx index 2e6ba3cd2..b5642ceba 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx @@ -61,10 +61,17 @@ describe('CNPGBackupSummary', () => { expect(t).toContain('main-2') expect(t).toContain('can not upload') expect(t).toContain('ScheduledBackup nightly') - expect(t).toContain('ObjectStore store') + expect(t).toContain("ObjectStore store · from the Cluster's current configuration") expect(t).not.toContain('On demand') }) + it('does not link a same-name schedule that replaced the one that took the backup', () => { + const owned = { ...b, metadata: { ...b.metadata, ownerReferences: [{ apiVersion: PG, kind: 'ScheduledBackup', name: 'nightly', uid: 'old' }] } } + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg', uid: 'new' } } + const t = text(renderToString()) + expect(t).toContain('an earlier schedule of that name') + }) + it('shows its own issues on top', () => { const issues = [ { id: 'i1', severity: 'critical' as const, kind: 'Backup', group: 'postgresql.cnpg.io', namespace: 'pg', name: 'main-20260901', reason: 'CNPGBackupFailed', message: 'Backup failed' }, @@ -81,6 +88,9 @@ describe('CNPGScheduledBackupSummary', () => { const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' }, spec: { cluster: { name: 'main' }, schedule: '0 0 0 * * *' } } const run = { apiVersion: PG, kind: 'Backup', metadata: { name: 'run-1', namespace: 'pg', labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, status: { phase: 'completed', startedAt: '2026-09-01T00:00:00Z' } } const t = text(renderToString()) + expect(t).toContain('0 0 0 * * *') + expect(t).toContain('seconds first') + expect(t).not.toContain('Daily') expect(t).toContain('run-1') expect(t).toContain('Completed') expect(t).toContain('Backups older than 7 days are not listed') @@ -118,7 +128,8 @@ describe('CNPGObjectStoreSummary', () => { expect(t).toContain('inferred') expect(t).toContain("Inferred from 1 cluster's WAL archiving and backup results — ObjectStore has no health status") expect(t).toContain('Uploads failing') - expect(t).toContain('The window is still restorable up to the last successful backup; it stops advancing while uploads fail.') + expect(t).toContain('A backup failed after the last recorded success.') + expect(t).not.toContain('restorable') expect(t).toContain('s3-creds') expect(t).toContain('s3://bucket/pg') }) @@ -129,6 +140,13 @@ describe('CNPGObjectStoreSummary', () => { expect(html).not.toContain('SECRET') }) + it('records a failure with no success without claiming a recovery point', () => { + const failing = { ...store, status: { serverRecoveryWindow: { main: { lastFailedBackupTime: '2026-09-03T00:00:00Z' } } } } + const t = text(renderToString()) + expect(t).toContain('No successful backup recorded.') + expect(t).not.toContain('restorable') + }) + it('says when no visible cluster uses the store', () => { const t = text(renderToString()) expect(t).toContain('No visible cluster uses this store') @@ -150,7 +168,8 @@ describe('CNPGDatabaseSummary', () => { expect(t).toContain('Pending') expect(t).not.toContain('Not applied') expect(t).toContain('drops it from PostgreSQL') - expect(t).toContain('Applied directly (no GitOps owner label)') + expect(t).toContain('GitOps source not recorded') + expect(t).not.toContain('Applied directly') }) it('reports a failure and the missing managed role beside it', () => { @@ -190,4 +209,23 @@ describe('CNPGImageCatalogSummary', () => { expect(t).toContain('Requests PostgreSQL 17 · not in this catalog') expect(t).toContain('clusters in namespaces you cannot read are not listed') }) + + it('keeps known users when cluster coverage is partial', () => { + const catalog = { apiVersion: PG, kind: 'ClusterImageCatalog', metadata: { name: 'pg' }, spec: { images: [{ major: 16, image: 'img:16' }] } } + const clusters = [{ apiVersion: PG, kind: 'Cluster', metadata: { name: 'a', namespace: 'x' }, spec: { imageCatalogRef: { kind: 'ClusterImageCatalog', name: 'pg', major: 16 } } }] + const t = text( + renderToString( + , + ), + ) + expect(t).toContain('x/a') + expect(t).toContain('Only clusters in namespaces you can read are listed.') + expect(t).not.toContain('No access to Clusters') + }) + + it('reserves the unavailable text for denied coverage', () => { + const catalog = { apiVersion: PG, kind: 'ClusterImageCatalog', metadata: { name: 'pg' }, spec: {} } + const t = text(renderToString()) + expect(t).toContain('No access to Clusters') + }) }) diff --git a/packages/k8s-ui/src/components/cnpg/relations.test.ts b/packages/k8s-ui/src/components/cnpg/relations.test.ts index 5d136500e..437dafbc7 100644 --- a/packages/k8s-ui/src/components/cnpg/relations.test.ts +++ b/packages/k8s-ui/src/components/cnpg/relations.test.ts @@ -7,6 +7,7 @@ import { databaseForDeclaration, gitopsSourceOf, inferredObjectStoreHealth, + isBackupFromSchedule, issuesForObject, missingManagedRole, objectStoreForBackup, @@ -46,11 +47,15 @@ function backup(name: string, extra: any = {}): any { } describe('scheduledBackupOf / backupsForScheduledBackup', () => { - const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' } } + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg', uid: 'uid-new' } } + const ownedBy = (name: string, uid: string, extra: any = {}) => + backup(name, { + ...extra, + metadata: { ...(extra.metadata ?? {}), ownerReferences: [{ apiVersion: PG, kind: 'ScheduledBackup', name: 'nightly', uid }] }, + }) - it('reads the owner reference', () => { - const b = backup('b1', { metadata: { ownerReferences: [{ apiVersion: PG, kind: 'ScheduledBackup', name: 'nightly' }] } }) - expect(scheduledBackupOf(b)).toBe('nightly') + it('reads the owner reference name', () => { + expect(scheduledBackupOf(ownedBy('b1', 'uid-new'))).toBe('nightly') }) it('falls back to the operator label when the owner is the Cluster or unset', () => { @@ -61,6 +66,7 @@ describe('scheduledBackupOf / backupsForScheduledBackup', () => { }, }) expect(scheduledBackupOf(b)).toBe('nightly') + expect(isBackupFromSchedule(b, sched)).toBe(true) }) it('returns null for an on-demand backup', () => { @@ -72,6 +78,12 @@ describe('scheduledBackupOf / backupsForScheduledBackup', () => { expect(scheduledBackupOf(b)).toBeNull() }) + it('does not attribute a recreated schedule the old schedule\'s Backups', () => { + const old = ownedBy('old', 'uid-old', { metadata: { labels: { 'cnpg.io/scheduled-backup': 'nightly' } } }) + expect(isBackupFromSchedule(old, sched)).toBe(false) + expect(isBackupFromSchedule(ownedBy('cur', 'uid-new'), sched)).toBe(true) + }) + it('lists owned backups newest first and never a Velero Backup', () => { const owned = (name: string, startedAt: string, apiVersion = PG) => ({ ...backup(name, { metadata: { labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, status: { startedAt } }), @@ -81,6 +93,7 @@ describe('scheduledBackupOf / backupsForScheduledBackup', () => { owned('old', '2026-09-01T00:00:00Z'), owned('new', '2026-09-02T00:00:00Z'), owned('velero', '2026-09-03T00:00:00Z', VELERO), + ownedBy('stale', 'uid-old', { status: { startedAt: '2026-09-04T00:00:00Z' } }), backup('other'), { ...owned('elsewhere', '2026-09-03T00:00:00Z'), metadata: { name: 'elsewhere', namespace: 'x', labels: { 'cnpg.io/scheduled-backup': 'nightly' } } }, ] @@ -91,20 +104,22 @@ describe('scheduledBackupOf / backupsForScheduledBackup', () => { describe('objectStoreForBackup / backupDestination', () => { const clusters = [pluginCluster('main', 'store-a')] - it('prefers the backup plugin parameters', () => { + it('prefers the store the backup recorded', () => { const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN, parameters: { barmanObjectName: 'store-b' } } } }) - expect(objectStoreForBackup(b, clusters)).toBe('store-b') + expect(objectStoreForBackup(b, clusters)).toEqual({ name: 'store-b', inferred: false }) }) - it('falls back to the target cluster plugin', () => { + it("marks a store taken from the Cluster's current plugin as inferred", () => { const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN } } }) - expect(objectStoreForBackup(b, clusters)).toBe('store-a') - expect(backupDestination(b, clusters)).toEqual({ type: 'objectStore', name: 'store-a' }) + expect(objectStoreForBackup(b, clusters)).toEqual({ name: 'store-a', inferred: true }) + expect(backupDestination(b, clusters)).toEqual({ type: 'objectStore', name: 'store-a', inferred: true }) }) - it('does not attribute another plugin to the barman store', () => { - const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: 'other.example.com' } } }) - expect(objectStoreForBackup(b, clusters)).toBeNull() + it('checks plugin identity before reading barmanObjectName', () => { + const other = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: 'other.example.com', parameters: { barmanObjectName: 'store-b' } } } }) + expect(objectStoreForBackup(other, clusters)).toBeNull() + const unnamed = backup('b', { spec: { method: 'plugin', pluginConfiguration: { parameters: { barmanObjectName: 'store-b' } } } }) + expect(objectStoreForBackup(unnamed, clusters)).toBeNull() }) it('reports in-tree paths and volume snapshots', () => { diff --git a/packages/k8s-ui/src/components/cnpg/relations.ts b/packages/k8s-ui/src/components/cnpg/relations.ts index cc840ddbd..2db316d66 100644 --- a/packages/k8s-ui/src/components/cnpg/relations.ts +++ b/packages/k8s-ui/src/components/cnpg/relations.ts @@ -151,28 +151,44 @@ export function problemsForObject(issues: CNPGWorkspaceIssue[] | undefined, ref: // Backups and schedules // --------------------------------------------------------------------------- +function scheduleOwnerRefs(backup: any): any[] { + const refs = backup?.metadata?.ownerReferences + if (!Array.isArray(refs)) return [] + return refs.filter((r: any) => r?.kind === 'ScheduledBackup' && isApiGroup(r?.apiVersion, CNPG_GROUP)) +} + /** - * The ScheduledBackup that created a Backup. The owner reference is only set - * when the schedule's `backupOwnerReference` is `self`; the operator labels - * every Backup it creates from a schedule regardless, so the label is read too. + * The name of the ScheduledBackup that created a Backup. The owner reference + * is only set when the schedule's `backupOwnerReference` is `self`; the + * operator labels every Backup it creates from a schedule regardless, so the + * label is the fallback when no such owner reference exists. */ export function scheduledBackupOf(backup: any): string | null { - const refs = backup?.metadata?.ownerReferences - if (Array.isArray(refs)) { - const owner = refs.find((r: any) => r?.kind === 'ScheduledBackup' && isApiGroup(r?.apiVersion, CNPG_GROUP)) - if (owner?.name) return owner.name - } + const owner = scheduleOwnerRefs(backup)[0] + if (owner?.name) return owner.name const label = backup?.metadata?.labels?.[SCHEDULED_BACKUP_LABEL] return typeof label === 'string' && label ? label : null } +/** + * Whether this schedule created the Backup. An owner reference must match by + * uid: a schedule deleted and recreated under the same name did not create the + * old one's Backups. The label, which carries only a name, is read only when + * no ScheduledBackup owner reference exists. + */ +export function isBackupFromSchedule(backup: any, schedule: any): boolean { + if (!isCNPGKind(backup, 'Backup') || nsOf(backup) !== nsOf(schedule)) return false + const owners = scheduleOwnerRefs(backup) + if (owners.length > 0) { + const uid = schedule?.metadata?.uid + return owners.some((r: any) => r?.name === nameOf(schedule) && !!uid && r?.uid === uid) + } + return backup?.metadata?.labels?.[SCHEDULED_BACKUP_LABEL] === nameOf(schedule) +} + /** Backups a ScheduledBackup created, newest first. */ export function backupsForScheduledBackup(schedule: any, backups: any[]): any[] { - const ns = nsOf(schedule) - const name = nameOf(schedule) - return backups - .filter((b) => isCNPGKind(b, 'Backup') && nsOf(b) === ns && scheduledBackupOf(b) === name) - .sort((a, b) => backupTime(b) - backupTime(a)) + return backups.filter((b) => isBackupFromSchedule(b, schedule)).sort((a, b) => backupTime(b) - backupTime(a)) } export function backupTime(backup: any): number { @@ -180,29 +196,32 @@ export function backupTime(backup: any): number { } /** - * The ObjectStore a plugin Backup wrote to: its own plugin parameters first, - * then the target Cluster's barman-cloud plugin. Null for non-plugin methods. + * The ObjectStore a barman-cloud plugin Backup wrote to. The Backup's own + * plugin parameters are a record of that run; the target Cluster's plugin + * configuration is only what it is configured with now, so a store taken from + * there is marked inferred. */ -export function objectStoreForBackup(backup: any, clusters: any[]): string | null { +export function objectStoreForBackup(backup: any, clusters: any[]): { name: string; inferred: boolean } | null { const method = backup?.status?.method || backup?.spec?.method if (method !== 'plugin') return null const cfg = backup?.spec?.pluginConfiguration + if (cfg?.name !== CNPG_BARMAN_PLUGIN_NAME) return null const own = cfg?.parameters?.barmanObjectName - if (typeof own === 'string' && own) return own - if (cfg?.name && cfg.name !== CNPG_BARMAN_PLUGIN_NAME) return null + if (typeof own === 'string' && own) return { name: own, inferred: false } const cluster = targetCluster(backup, clusters) - return (cluster && getCNPGClusterBarmanPlugin(cluster)?.barmanObjectName) || null + const current = cluster ? getCNPGClusterBarmanPlugin(cluster)?.barmanObjectName : undefined + return current ? { name: current, inferred: true } : null } export type CNPGBackupDestination = - | { type: 'objectStore'; name: string } + | { type: 'objectStore'; name: string; inferred: boolean } | { type: 'path'; path: string } | { type: 'volumeSnapshot' } | { type: 'unknown' } export function backupDestination(backup: any, clusters: any[]): CNPGBackupDestination { const store = objectStoreForBackup(backup, clusters) - if (store) return { type: 'objectStore', name: store } + if (store) return { type: 'objectStore', ...store } const method = backup?.status?.method || backup?.spec?.method if (method === 'volumeSnapshot') return { type: 'volumeSnapshot' } const path = backup?.status?.destinationPath diff --git a/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx b/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx index 2f8fe643b..d38b06858 100644 --- a/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx +++ b/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx @@ -61,9 +61,15 @@ export interface WorkloadLogsViewerProps { * re-armed. Requires `createStream`. Default: false. */ autoStream?: boolean + /** Pods selected when the pod list first loads; all pods when empty or none match. */ + initialPods?: string[] } -export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownload, forceDark, autoStream = false }: WorkloadLogsViewerProps) { +export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownload, forceDark, autoStream = false, initialPods }: WorkloadLogsViewerProps) { + const initialSelection = (names: string[]) => { + const wanted = names.filter((n) => initialPods?.includes(n)) + return new Set(wanted.length > 0 ? wanted : names) + } const [selectedContainer, setSelectedContainer] = useState('') const [pods, setPods] = useState([]) const [selectedPods, setSelectedPods] = useState>(new Set()) @@ -127,7 +133,9 @@ export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownl const previousPods = previousSnapshotPods.current const nextPods = resultPods.map(p => p.name) previousSnapshotPods.current = nextPods - setSelectedPods(selected => previousPods === null || previousPods.every(pod => selected.has(pod)) + setSelectedPods(selected => previousPods === null + ? initialSelection(nextPods) + : previousPods.every(pod => selected.has(pod)) ? new Set(nextPods) : new Set(nextPods.filter(pod => selected.has(pod)))) @@ -192,7 +200,7 @@ export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownl setEmptyMessage(data.emptyMessage || null) setEmptyCommand(data.command || null) setSelectedPods(prev => ( - prev.size === 0 ? new Set(nextPods.map((p: WorkloadPodInfo) => p.name)) : prev + prev.size === 0 ? initialSelection(nextPods.map((p: WorkloadPodInfo) => p.name)) : prev )) } }, diff --git a/packages/k8s-ui/src/components/workload/WorkloadView.tsx b/packages/k8s-ui/src/components/workload/WorkloadView.tsx index 248d635a6..0e0805d8d 100644 --- a/packages/k8s-ui/src/components/workload/WorkloadView.tsx +++ b/packages/k8s-ui/src/components/workload/WorkloadView.tsx @@ -80,6 +80,16 @@ import { isCoreBatchJob } from '../../utils/api-resources' export type WorkloadTabType = 'overview' | 'spec' | 'topology' | 'timeline' | 'logs' | 'metrics' | 'reachability' | 'cost' | 'yaml' type TabType = WorkloadTabType +/** A host-provided tab. `after` places it behind a built-in tab; `replaces` hides that built-in tab. */ +export interface WorkloadExtraTab { + id: string + label: string + icon?: ReactNode + after?: WorkloadTabType + replaces?: WorkloadTabType + render: () => ReactNode +} + export interface ResourceOwnershipContext { application?: { key: string @@ -302,6 +312,8 @@ interface WorkloadViewProps { context: 'drawer' | 'expanded' onNavigate?: NavigateToResource }) => ReactNode + /** Extra tabs for the expanded view (e.g. a domain's own sections). */ + extraTabs?: WorkloadExtraTab[] /** Render a full replacement for the expanded Overview tab. */ renderExpandedOverview?: (props: { kind: string @@ -451,6 +463,7 @@ export function WorkloadView({ reachableVia, renderExpandedOverview, renderSummary, + extraTabs, renderRelatedYaml, renderMetricsTab, renderCostTab, @@ -819,8 +832,10 @@ export function WorkloadView({ { id: 'cost', label: 'Cost', icon: , hidden: !costTabVisible }, { id: 'yaml', label: 'YAML', icon: }, ] - const requestedTabAvailable = tabs.some((tab) => tab.id === requestedTab && !tab.hidden) + const allTabs = mergeExtraTabs(tabs, expanded ? extraTabs : undefined) + const requestedTabAvailable = allTabs.some((tab) => tab.id === requestedTab && !tab.hidden) const effectiveTab: TabType = requestedTabAvailable ? requestedTab : 'overview' + const activeExtraTab = expanded ? extraTabs?.find((x) => x.id === effectiveTab) : undefined const shouldCommitFallback = requestedTab !== 'overview' && !requestedTabAvailable && @@ -1138,7 +1153,7 @@ export function WorkloadView({ )} } - tabs={tabs} + tabs={allTabs} activeTab={effectiveTab} onTabChange={handleSetTab} scopeControls={scopeControls} @@ -1153,6 +1168,7 @@ export function WorkloadView({
)}
+ {activeExtraTab &&
{activeExtraTab.render()}
} {effectiveTab === 'overview' && expandedSummary ? (
{expandedSummary}
) : effectiveTab === 'overview' && expandedOverview ? ( @@ -3890,3 +3906,17 @@ function mergeAndRankEvents(events: TimelineEvent[], updates: TimelineEvent[]): return new Date(b.timestamp).getTime() - new Date(a.timestamp).getTime() }) } + +function mergeExtraTabs(tabs: DetailShellTab[], extra: WorkloadExtraTab[] | undefined): DetailShellTab[] { + if (!extra || extra.length === 0) return tabs + const replaced = new Set(extra.map((x) => x.replaces).filter(Boolean)) + const out = tabs.map((t) => (replaced.has(t.id) ? { ...t, hidden: true } : t)) + for (const x of extra) { + const tab: DetailShellTab = { id: x.id as TabType, label: x.label, icon: x.icon } + const anchor = x.after ?? x.replaces + const idx = anchor ? out.findIndex((t) => t.id === anchor) : -1 + if (idx >= 0) out.splice(idx + 1, 0, tab) + else out.push(tab) + } + return out +} diff --git a/packages/k8s-ui/src/components/workload/index.ts b/packages/k8s-ui/src/components/workload/index.ts index 1d71d59b1..693176b3b 100644 --- a/packages/k8s-ui/src/components/workload/index.ts +++ b/packages/k8s-ui/src/components/workload/index.ts @@ -8,5 +8,6 @@ export { type ResourceOwnershipContext, type ServingResourceDetail, type WorkloadTabType, + type WorkloadExtraTab, } from './WorkloadView' export { ResourceDetailDrawer } from './ResourceDetailDrawer' diff --git a/pkg/timeline/converter.go b/pkg/timeline/converter.go index c59ae8311..50ca141d9 100644 --- a/pkg/timeline/converter.go +++ b/pkg/timeline/converter.go @@ -9,6 +9,7 @@ import ( corev1 "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" ) // informerEventID derives a deterministic id from the resource's observable @@ -183,9 +184,6 @@ func ExtractLabels(obj any) map[string]string { } allLabels := meta.GetLabels() - if len(allLabels) == 0 { - return nil - } // Only keep labels that are useful for grouping. The GitOps identity // labels must ride along or the app-membership matchKeys the server ships @@ -214,6 +212,9 @@ func ExtractLabels(obj any) map[string]string { relevant[key] = v } } + if cluster := cnpgOwningCluster(obj, allLabels); cluster != "" { + relevant[CNPGClusterLabel] = cluster + } if len(relevant) == 0 { return nil @@ -221,6 +222,35 @@ func ExtractLabels(obj any) map[string]string { return relevant } +// CNPGClusterLabel is the retained label naming the CloudNativePG Cluster a +// row's subject belongs to. +const CNPGClusterLabel = "cnpg.io/cluster" + +// cnpgOwningCluster names the CloudNativePG Cluster an object belongs to, so a +// child's rows stay attributable after it is deleted. CNPG children that +// reference their Cluster through spec.cluster.name (Backup, Pooler, Database, +// ...) are not all labelled; the operator only labels what it creates. +func cnpgOwningCluster(obj any, labels map[string]string) string { + switch o := obj.(type) { + case *corev1.Pod: + return labels[CNPGClusterLabel] + case *unstructured.Unstructured: + group := resourceid.GroupFromAPIVersion(o.GetAPIVersion()) + if group == "" && o.GetKind() == "Pod" { + return labels[CNPGClusterLabel] + } + if group != "postgresql.cnpg.io" && group != "barmancloud.cnpg.io" { + return "" + } + if v := labels[CNPGClusterLabel]; v != "" { + return v + } + name, _, _ := unstructured.NestedString(o.Object, "spec", "cluster", "name") + return name + } + return "" +} + // Resource health classification for timeline events lives with the canonical // classifiers in internal/k8s (classifyTimelineHealth → ClassifyPodHealth), not // here: the timeline package can't reach that logic across the module boundary, diff --git a/pkg/timeline/converter_cnpg_test.go b/pkg/timeline/converter_cnpg_test.go new file mode 100644 index 000000000..47338170c --- /dev/null +++ b/pkg/timeline/converter_cnpg_test.go @@ -0,0 +1,66 @@ +package timeline + +import ( + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" +) + +func cnpgTestObject(apiVersion, kind string, labels map[string]string, spec map[string]any) *unstructured.Unstructured { + u := &unstructured.Unstructured{Object: map[string]any{"apiVersion": apiVersion, "kind": kind, "metadata": map[string]any{"name": "x", "namespace": "db"}}} + if spec != nil { + u.Object["spec"] = spec + } + if labels != nil { + u.SetLabels(labels) + } + return u +} + +func TestExtractLabelsRetainsTheOwningCNPGCluster(t *testing.T) { + clusterRef := map[string]any{"cluster": map[string]any{"name": "pg-orders"}} + cases := []struct { + name string + obj any + want string + }{ + {"unlabelled Backup records spec.cluster.name", cnpgTestObject("postgresql.cnpg.io/v1", "Backup", nil, clusterRef), "pg-orders"}, + {"every spec.cluster kind", cnpgTestObject("postgresql.cnpg.io/v1", "Database", map[string]string{"team": "a"}, clusterRef), "pg-orders"}, + {"the label wins over spec", cnpgTestObject("postgresql.cnpg.io/v1", "Pooler", map[string]string{"cnpg.io/cluster": "pg-labelled"}, clusterRef), "pg-labelled"}, + {"barman ObjectStore label", cnpgTestObject("barmancloud.cnpg.io/v1", "ObjectStore", map[string]string{"cnpg.io/cluster": "pg-orders"}, nil), "pg-orders"}, + {"a Velero Backup is not attributed", cnpgTestObject("velero.io/v1", "Backup", map[string]string{"cnpg.io/cluster": "pg-orders"}, clusterRef), ""}, + {"a Deployment keeps no cnpg label", cnpgTestObject("apps/v1", "Deployment", map[string]string{"cnpg.io/cluster": "pg-orders"}, nil), ""}, + {"CNPG object without a cluster", cnpgTestObject("postgresql.cnpg.io/v1", "ImageCatalog", nil, nil), ""}, + {"typed Pod", &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "pg-orders-1", Labels: map[string]string{"cnpg.io/cluster": "pg-orders"}}}, "pg-orders"}, + {"unstructured Pod", cnpgTestObject("v1", "Pod", map[string]string{"cnpg.io/cluster": "pg-orders"}, nil), "pg-orders"}, + {"a Pod's spec is never read", cnpgTestObject("v1", "Pod", nil, clusterRef), ""}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + got := ExtractLabels(tc.obj)[CNPGClusterLabel] + if got != tc.want { + t.Fatalf("cnpg.io/cluster = %q, want %q (labels %v)", got, tc.want, ExtractLabels(tc.obj)) + } + }) + } +} + +func TestExtractLabelsStaysNilWithoutAnythingToRetain(t *testing.T) { + if got := ExtractLabels(cnpgTestObject("apps/v1", "Deployment", nil, nil)); got != nil { + t.Fatalf("labels = %v, want nil", got) + } + if got := ExtractLabels(cnpgTestObject("apps/v1", "Deployment", map[string]string{"unrelated": "x"}, nil)); got != nil { + t.Fatalf("labels = %v, want nil", got) + } +} + +// A deleted child's tombstone keeps the attribution, so K8s Events that arrive +// after the delete still name the Cluster. +func TestTombstoneEntryCarriesCNPGAttribution(t *testing.T) { + entry, ok := ExtractTombstoneEntry(cnpgTestObject("postgresql.cnpg.io/v1", "Backup", nil, map[string]any{"cluster": map[string]any{"name": "pg-orders"}})) + if !ok || entry.Labels[CNPGClusterLabel] != "pg-orders" { + t.Fatalf("tombstone labels = %v ok=%v", entry.Labels, ok) + } +} diff --git a/web/src/App.tsx b/web/src/App.tsx index f93779647..1ed06998d 100644 --- a/web/src/App.tsx +++ b/web/src/App.tsx @@ -39,6 +39,8 @@ import { useNavCustomization } from './context/NavCustomization' import type { FleetTakeoverTarget } from './context/NavCustomization' import { PrimaryNavRail } from './components/nav/PrimaryNavRail' import { CNPGView } from './components/cnpg/CNPGView' +import { CNPG_SCREENS, cnpgDetailKindFor, cnpgDetailPath, parseCNPGRoute } from './components/cnpg/routes' +import { currentPageLabel } from './components/cnpg/paths' import { navigateFromPrimaryRail } from './components/nav/navigation' import { useNavRailPinned } from './hooks/useNavRailPinned' import { useMediaQuery } from './hooks/useMediaQuery' @@ -283,7 +285,12 @@ function radarPageTitle(pathname: string, search = '', apiResources?: APIResourc if (pathSegments[1] === 'activity') return 'Capacity Activity' } - if (view === 'cnpg') return 'CloudNativePG' + if (view === 'cnpg') { + const route = parseCNPGRoute(pathname) + if (route.detail) return route.detail.name + const screen = CNPG_SCREENS.find((s) => s.id === route.screen) + return `CloudNativePG ${screen?.label ?? 'Overview'}` + } if (view === 'home') return 'Overview' // Every other view's label is its id capitalized — getViewFromPath has already // normalized aliases (e.g. /audit → 'checks'), so no lookup table is needed. @@ -1232,6 +1239,10 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL const nextParams = new URLSearchParams() const diagnoseRun = new URLSearchParams(location.search).get('ai-run') if (diagnoseRun) nextParams.set('ai-run', diagnoseRun) + // A CNPG detail keeps the context it belongs to, so it can say it is + // not in the new one instead of loading a same-named object. + const pinnedCtx = new URLSearchParams(location.search).get('ctx') + if (pinnedCtx && location.pathname.startsWith('/cnpg/')) nextParams.set('ctx', pinnedCtx) navigate( { pathname: location.pathname, search: nextParams.toString() }, { replace: true, state: location.state }, @@ -2387,7 +2398,15 @@ function AppInner({ manageDocumentTitle = false, documentTitleSuffix, onClusterL onClose={closeDrawer} onNavigate={(res) => navigateToResource(res)} canCollapseToDrawer={!isMobile} - onExpand={(_res, opts) => { + onExpand={(res, opts) => { + const cnpgPlural = cnpgDetailKindFor(res.kind, res.group) + if (cnpgPlural) { + navigate( + cnpgDetailPath({ plural: cnpgPlural, namespace: res.namespace, name: res.name }, connection.context || undefined, opts?.yaml ? 'yaml' : undefined), + { state: { returnLabel: currentPageLabel(), returnCtx: connection.context } }, + ) + return + } // Grow the peek into a fullscreen overlay (?full=1, pushed so Back // collapses) over whatever view is underneath — list, topology graph, // GitOps, Applications — which stays mounted. Carry the YAML tab when diff --git a/web/src/api/cnpg.ts b/web/src/api/cnpg.ts index 5ccace5e3..89a560edf 100644 --- a/web/src/api/cnpg.ts +++ b/web/src/api/cnpg.ts @@ -1,5 +1,5 @@ import { useQuery } from '@tanstack/react-query' -import type { CNPGWorkspaceResponse } from '@skyhook-io/k8s-ui' +import type { CNPGWorkspaceResponse, TimelineEvent } from '@skyhook-io/k8s-ui' import { fetchJSON } from './client' // /api/cnpg/workspace @@ -67,3 +67,29 @@ export function useCNPGOperator(options?: { enabled?: boolean }) { refetchInterval: 60_000, }) } + +export interface CNPGClusterActivityResponse { + events: TimelineEvent[] + oldest: string | null + attributionSince: string | null + truncated: boolean +} + +// /api/cnpg/clusters/{ns}/{name}/activity +// +// The Cluster's history together with its instance Pods and every CNPG object +// attributed to it, including ones since deleted. +export function useCNPGClusterActivity(namespace: string, name: string, sinceHours = 24) { + return useQuery({ + queryKey: ['cnpg', 'activity', namespace, name, sinceHours], + queryFn: ({ signal }) => { + const since = new Date(Date.now() - sinceHours * 3600_000).toISOString() + return fetchJSON( + `/cnpg/clusters/${encodeURIComponent(namespace)}/${encodeURIComponent(name)}/activity?since=${encodeURIComponent(since)}&limit=500`, + signal, + ) + }, + staleTime: 10_000, + refetchInterval: 30_000, + }) +} diff --git a/web/src/components/cnpg/CNPGClusterActivity.tsx b/web/src/components/cnpg/CNPGClusterActivity.tsx new file mode 100644 index 000000000..80af1506b --- /dev/null +++ b/web/src/components/cnpg/CNPGClusterActivity.tsx @@ -0,0 +1,48 @@ +import { useState } from 'react' +import { PaneLoader, TimelineList, formatAge, type NavigateToResource } from '@skyhook-io/k8s-ui' +import { useCNPGClusterActivity } from '../../api/cnpg' +import { Notice } from '../capacity/shared' +import { Segments } from './shared' + +const RANGES = [ + { id: '6', label: '6 h' }, + { id: '24', label: '24 h' }, + { id: '168', label: '7 d' }, +] as const + +/** + * Kubernetes events and changes for the Cluster, its instance Pods and the + * CNPG objects attributed to it — including Backups and declarations deleted + * since. Attribution of child objects relies on a label Radar records at + * ingestion, so history older than that is marked incomplete. + */ +export function CNPGClusterActivity({ namespace, name, onNavigate }: { namespace: string; name: string; onNavigate?: NavigateToResource }) { + const [hours, setHours] = useState('24') + const q = useCNPGClusterActivity(namespace, name, Number(hours)) + + return ( +
+
+ ({ id: r.id, label: r.label }))} /> + + Kubernetes events and changes for the Cluster, its instances, Backups, Poolers and declarations. + +
+ {q.data?.attributionSince && ( +
+ Child-object history is complete since {formatAge(q.data.attributionSince)} ago; earlier Backups and declarations may be missing. +
+ )} + {q.data?.truncated && Showing the most recent events only; narrow the range to see all of them.} + {q.error && !q.data ? ( + Activity could not be loaded: {q.error instanceof Error ? q.error.message : 'unknown error'} + ) : !q.data ? ( + + ) : ( +
+ +
+ )} +
+ ) +} diff --git a/web/src/components/cnpg/CNPGClusterLogs.tsx b/web/src/components/cnpg/CNPGClusterLogs.tsx new file mode 100644 index 000000000..75a060534 --- /dev/null +++ b/web/src/components/cnpg/CNPGClusterLogs.tsx @@ -0,0 +1,59 @@ +import { useCallback } from 'react' +import { useSearchParams } from 'react-router-dom' +import { WorkloadLogsViewer, type WorkloadLogsFetchParams, type WorkloadLogsResult } from '@skyhook-io/k8s-ui' +import { fetchJSON } from '../../api/client' +import { getApiBase, getCredentialsMode } from '../../api/config' +import { useDesktopDownload } from '../../hooks/useDesktopDownload' +import { useTheme } from '../../context/ThemeContext' + +function logsPath(namespace: string, name: string) { + return `/cnpg/clusters/${encodeURIComponent(namespace)}/${encodeURIComponent(name)}/logs` +} + +function query(params: WorkloadLogsFetchParams, tailDefault?: number) { + const q = new URLSearchParams() + if (params.container) q.set('container', params.container) + const tail = params.tailLines ?? tailDefault + if (tail) q.set('tailLines', String(tail)) + if (params.sinceSeconds) q.set('sinceSeconds', String(params.sinceSeconds)) + const s = q.toString() + return s ? `?${s}` : '' +} + +/** + * Logs merged from every instance Pod of a CloudNativePG Cluster. `?pod=` + * preselects one instance (the fleet's "Logs" action and "Open instance logs"). + */ +export function CNPGClusterLogs({ namespace, name }: { namespace: string; name: string }) { + const [searchParams] = useSearchParams() + const pod = searchParams.get('pod') + const desktopDownload = useDesktopDownload() + const { theme } = useTheme() + + const fetchAll = useCallback( + (params: WorkloadLogsFetchParams) => + fetchJSON(`${logsPath(namespace, name)}${query(params)}`, { signal: params.signal }), + [namespace, name], + ) + const createStream = useCallback( + (params: WorkloadLogsFetchParams) => + new EventSource(`${getApiBase()}${logsPath(namespace, name)}/stream${query(params, 50)}`, { + withCredentials: getCredentialsMode() === 'include', + }), + [namespace, name], + ) + + return ( +
+ +
+ ) +} diff --git a/web/src/components/cnpg/CNPGDeclarations.tsx b/web/src/components/cnpg/CNPGDeclarations.tsx index 254db48bf..fd14248ba 100644 --- a/web/src/components/cnpg/CNPGDeclarations.tsx +++ b/web/src/components/cnpg/CNPGDeclarations.tsx @@ -1,7 +1,7 @@ import { useMemo, type ReactNode } from 'react' import { clsx } from 'clsx' import { AlertTriangle } from 'lucide-react' -import { Badge, isApiGroup, toneTextClass, type CNPGFleetRow } from '@skyhook-io/k8s-ui' +import { Badge, cnpgGitOpsSource, isApiGroup, toneTextClass, type CNPGFleetRow } from '@skyhook-io/k8s-ui' import type { SelectedResource } from '../../types' import { CNPGWorkspaceHeader, @@ -12,6 +12,8 @@ import { Sub, clusterResource, cnpgResource, + coverageEmpty, + worstCoverage, namespaceChip, type CNPGScreenProps, } from './shared' @@ -47,11 +49,9 @@ function stateOf(obj: any): State { } function gitopsSource(obj: any): string | undefined { - const labels = obj?.metadata?.labels ?? {} - if (labels['argocd.argoproj.io/instance']) return `Argo CD ${labels['argocd.argoproj.io/instance']}` - const flux = labels['kustomize.toolkit.fluxcd.io/name'] || labels['helm.toolkit.fluxcd.io/name'] - if (flux) return `Flux ${flux}` - return undefined + const src = cnpgGitOpsSource(obj) + if (!src) return undefined + return `${src.tool === 'argocd' ? 'Argo CD' : 'Flux'} ${src.name}` } const STATE_BADGE: Record = { @@ -71,7 +71,7 @@ function roleState(cluster: any, role: string): { state: State; error?: string } export function CNPGDeclarations({ data, fleet, namespaces, searchParams, onSetParams, onInspect, inspected, onClearNamespaces }: CNPGScreenProps) { const clusterFilter = searchParams.get('cluster') - const onlyFailed = searchParams.get('show') === 'failed' + const show = (searchParams.get('show') as 'failed' | 'pending' | null) ?? null const groups = useMemo(() => { const byCluster = new Map() @@ -172,23 +172,25 @@ export function CNPGDeclarations({ data, fleet, namespaces, searchParams, onSetP } return [...byCluster.values()] .filter((g) => !clusterFilter || `${g.namespace}/${g.cluster}` === clusterFilter) - .map((g) => ({ ...g, items: onlyFailed ? g.items.filter((i) => i.state === 'failed') : g.items })) + .map((g) => ({ ...g, items: show ? g.items.filter((i) => i.state === show) : g.items })) .filter((g) => g.items.length > 0) .sort((a, b) => { const fa = a.items.some((i) => i.state === 'failed') ? 0 : 1 const fb = b.items.some((i) => i.state === 'failed') ? 0 : 1 return fa - fb || a.namespace.localeCompare(b.namespace) || a.cluster.localeCompare(b.cluster) }) - }, [data.objects.databases, data.objects.publications, data.objects.subscriptions, fleet.rows, clusterFilter, onlyFailed]) + }, [data.objects.databases, data.objects.publications, data.objects.subscriptions, fleet.rows, clusterFilter, show]) - const failedTotal = useMemo(() => { - let n = 0 - for (const k of ['databases', 'publications', 'subscriptions'] as const) { - n += (data.objects[k] ?? []).filter((o) => o?.status?.applied === false).length + const totals = useMemo(() => { + let failed = 0 + let pending = 0 + for (const r of fleet.rows) { + failed += r.declarations.failed + pending += r.declarations.pending } - for (const r of fleet.rows) n += Object.keys(r.cluster?.status?.managedRolesStatus?.cannotReconcile ?? {}).length - return n - }, [data.objects, fleet.rows]) + return { failed, pending } + }, [fleet.rows]) + const declCoverage = worstCoverage(data.coverage.databases, data.coverage.publications, data.coverage.subscriptions) const chips = [ ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), @@ -206,11 +208,12 @@ export function CNPGDeclarations({ data, fleet, namespaces, searchParams, onSetP
onSetParams({ show: id === 'failed' ? 'failed' : null })} + value={show ?? 'all'} + onChange={(id) => onSetParams({ show: id === 'all' ? null : id })} options={[ { id: 'all', label: 'All declarations' }, - { id: 'failed', label: 'Not reconciled', count: failedTotal }, + { id: 'failed', label: 'Not applied', count: totals.failed }, + { id: 'pending', label: 'Pending', count: totals.pending }, ]} />
@@ -218,7 +221,12 @@ export function CNPGDeclarations({ data, fleet, namespaces, searchParams, onSetP {groups.length === 0 ? (
- {onlyFailed ? 'Every declaration in this scope is reconciled.' : 'No declarations in this scope.'} + {show === 'failed' + ? 'No declaration in this scope is reported as not applied.' + : show === 'pending' + ? 'No declaration in this scope is waiting for the operator.' + : coverageEmpty(declCoverage, 'declarations')} + {show && declCoverage?.state !== 'full' ? ' Some declarations are not readable with your access.' : ''}
) : ( groups.map((g) => { @@ -296,7 +304,7 @@ function DeclarationRow({ item, active, onInspect }: { item: DeclItem; active: b {item.isField && field of the Cluster}
- {item.source ?? Applied directly (no GitOps owner label)} + {item.source ?? GitOps source not recorded}
) diff --git a/web/src/components/cnpg/CNPGDetailPage.tsx b/web/src/components/cnpg/CNPGDetailPage.tsx new file mode 100644 index 000000000..41272f77e --- /dev/null +++ b/web/src/components/cnpg/CNPGDetailPage.tsx @@ -0,0 +1,228 @@ +import { useCallback, useEffect, useMemo } from 'react' +import { useLocation, useNavigate, useSearchParams } from 'react-router-dom' +import { Activity, ArrowLeft, Database, ShieldCheck, Unplug } from 'lucide-react' +import type { WorkloadExtraTab } from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { useConnection } from '../../context/ConnectionContext' +import { useContexts } from '../../api/client' +import { useContextSwitchFlow } from '../useContextSwitchFlow' +import { WorkloadView } from '../workload/WorkloadView' +import { EmptyState } from '../capacity/shared' +import { CNPGClusterActivity } from './CNPGClusterActivity' +import { CNPGProtection } from './CNPGProtection' +import { CNPGScreenGate } from './shared' +import { CNPG_DETAIL_KINDS, CNPG_SCREENS, cnpgDetailKindFor, cnpgDetailPath, cnpgScreenPath, type CNPGDetailTarget } from './routes' +import { currentPageLabel } from './paths' +import { useCNPGFleet } from './useCNPGSidebarWorkspace' + +interface ReturnState { + returnLabel?: string + returnCtx?: string +} + +function ClusterProtectionTab({ namespace, name, onInspect }: { namespace: string; name: string; onInspect: (r: SelectedResource) => void }) { + const { query, fleet } = useCNPGFleet([namespace]) + const [searchParams] = useSearchParams() + return ( + + {(data, readyFleet) => ( + {}} + onInspect={onInspect} + inspected={null} + onClearNamespaces={() => {}} + scopeCluster={{ namespace, name }} + /> + )} + + ) +} + +/** + * The full detail of a CloudNativePG object, framed by the workspace: the + * Resources sidebar keeps the workspace destination highlighted, a return + * control goes back to the task the user came from, and the crumb names the + * object's place. The object's own sections come from Radar's detail view. + * + * `ctx` in the URL pins the Kubernetes context; when the active context is a + * different one, the page says so instead of loading a same-named object. + */ +export function CNPGDetailPage({ + target, + namespaces, + onOpenResource, +}: { + target: CNPGDetailTarget + namespaces: string[] + onOpenResource: (resource: SelectedResource) => void +}) { + const location = useLocation() + const navigate = useNavigate() + const [searchParams, setSearchParams] = useSearchParams() + const { connection } = useConnection() + const activeContext = connection.context + const pinnedContext = searchParams.get('ctx') + + // A link opened without a context belongs to the one active now; pin it so a + // later context switch shows "not in this context" instead of a same-named + // object from the other cluster. + useEffect(() => { + if (pinnedContext || !activeContext) return + const params = new URLSearchParams(searchParams) + params.set('ctx', activeContext) + setSearchParams(params, { replace: true, state: location.state }) + }, [pinnedContext, activeContext]) // eslint-disable-line react-hooks/exhaustive-deps + const returnState = (location.state ?? {}) as ReturnState + const spec = CNPG_DETAIL_KINDS[target.plural] + const home = CNPG_SCREENS.find((s) => s.id === spec.home)! + const returnLabel = returnState.returnLabel && (!returnState.returnCtx || returnState.returnCtx === activeContext) ? returnState.returnLabel : null + + const openRelated = useCallback( + (res: SelectedResource) => { + const plural = cnpgDetailKindFor(res.kind, res.group) + if (plural) { + navigate(cnpgDetailPath({ plural, namespace: res.namespace, name: res.name }, activeContext), { + state: { returnLabel: currentPageLabel(), returnCtx: activeContext } satisfies ReturnState, + }) + } else { + onOpenResource(res) + } + }, + [navigate, activeContext, onOpenResource], + ) + + const extraTabs = useMemo(() => { + if (target.plural !== 'clusters') return undefined + return [ + { + id: 'protection', + label: 'Protection', + icon: , + after: 'spec', + render: () => , + }, + { + id: 'activity', + label: 'Activity', + icon: , + replaces: 'timeline', + render: () => ( + + ), + }, + ] + }, [target.plural, target.namespace, target.name, onOpenResource, openRelated]) + + if (pinnedContext && activeContext && pinnedContext !== activeContext) { + return + } + + const outsideFilter = namespaces.length > 0 && !!target.namespace && !namespaces.includes(target.namespace) + + const breadcrumb = ( +
+ {returnLabel && ( + <> + + + + )} + + {outsideFilter && ( + + Namespace {target.namespace} is outside your namespace filter; this object stays open. + + )} +
+ ) + + return ( +
+ (returnLabel ? navigate(-1) : navigate(home.path))} + hideBackButton + breadcrumb={breadcrumb} + onNavigateToResource={openRelated} + extraTabs={extraTabs} + /> +
+ ) +} + +function NotInContext({ + target, + pinnedContext, + activeContext, + homeLabel, + homePath, +}: { + target: CNPGDetailTarget + pinnedContext: string + activeContext: string + homeLabel: string + homePath: string +}) { + const navigate = useNavigate() + const { data: contexts } = useContexts() + const { requestSwitch, confirmDialog } = useContextSwitchFlow() + const pinned = contexts?.find((c) => c.name === pinnedContext) + return ( + <> + + {pinned && ( + + )} + +
+ } + /> + {confirmDialog} + + ) +} diff --git a/web/src/components/cnpg/CNPGDrawerTrail.tsx b/web/src/components/cnpg/CNPGDrawerTrail.tsx new file mode 100644 index 000000000..58bdf2d51 --- /dev/null +++ b/web/src/components/cnpg/CNPGDrawerTrail.tsx @@ -0,0 +1,37 @@ +import { useLocation, useSearchParams } from 'react-router-dom' +import { ArrowLeft } from 'lucide-react' +import { CNPG_KIND_BY_KEY } from '@skyhook-io/k8s-ui' +import type { SelectedResource } from '../../types' +import { decodeDrawerTrail, encodeDrawerTrail, sameResource } from './routes' + +const KIND_BY_PLURAL: Record = Object.fromEntries( + Object.values(CNPG_KIND_BY_KEY).map((k) => [k.plural, k.kind]), +) + +/** + * On a CloudNativePG workspace screen the drawer URL carries the chain of + * objects opened from inside the drawer. This is the step back to the previous + * one, shown for whatever kind is open (a Pod or Secret reached from a CNPG + * object included). + */ +export function CNPGDrawerTrailBack({ resource }: { resource: SelectedResource }) { + const location = useLocation() + const [searchParams, setSearchParams] = useSearchParams() + if (!location.pathname.startsWith('/cnpg')) return null + const trail = decodeDrawerTrail(searchParams.get('drawer')) + if (trail.length < 2 || !sameResource(trail[trail.length - 1], resource)) return null + const prev = trail[trail.length - 2] + const back = () => { + const params = new URLSearchParams(searchParams) + params.set('drawer', encodeDrawerTrail(trail.slice(0, -1))) + setSearchParams(params, { replace: true }) + } + return ( +
+ +
+ ) +} diff --git a/web/src/components/cnpg/CNPGOperator.tsx b/web/src/components/cnpg/CNPGOperator.tsx index 2bf3e59e4..9f32d0e53 100644 --- a/web/src/components/cnpg/CNPGOperator.tsx +++ b/web/src/components/cnpg/CNPGOperator.tsx @@ -4,6 +4,9 @@ import { useCNPGOperator, type CNPGOperatorComponent, type CNPGOperatorConfig } import { Notice } from '../capacity/shared' import { CNPGWorkspaceHeader, + CoverageNotice, + coverageEmpty, + worstCoverage, Mono, ScreenBody, SectionTable, @@ -67,6 +70,7 @@ export function CNPGOperator({ data, fleet, onInspect, inspected }: CNPGScreenPr subtitle="Operator and plugin workloads, their versions, image catalogs and operator configuration." /> + {operator.isLoading && !op ? ( ) : !op ? ( @@ -159,7 +163,7 @@ export function CNPGOperator({ data, fleet, onInspect, inspected }: CNPGScreenPr rowResource={(c) => cnpgResource(c.kind === 'ImageCatalog' ? 'imagecatalogs' : 'clusterimagecatalogs', c.namespace, c.name)} onInspect={onInspect} inspected={inspected} - empty="No image catalogs in this scope." + empty={coverageEmpty(worstCoverage(data.coverage.imageCatalogs, data.coverage.clusterImageCatalogs), 'image catalogs')} footer={ <> Used-by lists only clusters you can see; the catalog detail asks the server for every user. diff --git a/web/src/components/cnpg/CNPGOverview.tsx b/web/src/components/cnpg/CNPGOverview.tsx index 977945b62..3937d0373 100644 --- a/web/src/components/cnpg/CNPGOverview.tsx +++ b/web/src/components/cnpg/CNPGOverview.tsx @@ -1,7 +1,7 @@ import { useMemo } from 'react' import { useNavigate } from 'react-router-dom' import { clsx } from 'clsx' -import { ArrowRight, Database, Search } from 'lucide-react' +import { ArrowRight, Database, FileText, Search } from 'lucide-react' import { CNPG_PROBLEM_CATEGORIES, FactValue, @@ -15,7 +15,7 @@ import type { SelectedResource } from '../../types' import { useConnection } from '../../context/ConnectionContext' import { EmptyState, ROW_HOVER, TABLE_HEAD, TABLE_WRAP, TBODY, TD, TH } from '../capacity/shared' import { CNPGWorkspaceHeader, CoverageNotice, FilterChips, type CNPGScreenProps } from './shared' -import { cnpgClusterFullPath } from './paths' +import { cnpgClusterFullPath, currentPageLabel } from './paths' import { sameResource } from './routes' type Filter = 'attention' | 'all' @@ -209,8 +209,8 @@ export function CNPGOverview({ - - + + @@ -252,16 +252,36 @@ export function CNPGOverview({ {row.pgVersion ?? '—'} - +
+ + +
) diff --git a/web/src/components/cnpg/CNPGPooling.tsx b/web/src/components/cnpg/CNPGPooling.tsx index fb18f07f0..21c42b50a 100644 --- a/web/src/components/cnpg/CNPGPooling.tsx +++ b/web/src/components/cnpg/CNPGPooling.tsx @@ -16,6 +16,7 @@ import { SectionTable, Sub, cnpgResource, + coverageEmpty, namespaceChip, type CNPGScreenProps, } from './shared' @@ -42,7 +43,6 @@ export function CNPGPooling({ data, fleet, namespaces, searchParams, onSetParams ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), ...namespaceChip(namespaces, onClearNamespaces), ] - const readable = data.coverage.poolers?.state === 'full' || data.coverage.poolers?.state === 'partial' return (
@@ -107,7 +107,7 @@ export function CNPGPooling({ data, fleet, namespaces, searchParams, onSetParams onInspect={onInspect} inspected={inspected} minWidth={880} - empty={readable ? 'No Poolers in this scope.' : 'Poolers are not readable with your access.'} + empty={coverageEmpty(data.coverage.poolers, 'Poolers')} footer="Instances are the Pooler’s own ready count. Client waits and server-pool saturation come from PgBouncer metrics, which Radar does not read yet." /> diff --git a/web/src/components/cnpg/CNPGProtection.tsx b/web/src/components/cnpg/CNPGProtection.tsx index e354b12a6..fe25b5d0b 100644 --- a/web/src/components/cnpg/CNPGProtection.tsx +++ b/web/src/components/cnpg/CNPGProtection.tsx @@ -2,7 +2,6 @@ import { useMemo } from 'react' import { Badge, FactValue, - cronToHuman, formatAge, formatDuration, getCNPGBackupStatus, @@ -24,6 +23,7 @@ import { SectionTable, Sub, clusterResource, + coverageEmpty, cnpgResource, namespaceChip, type CNPGScreenProps, @@ -78,8 +78,18 @@ function inferStoreHealth(store: any, users: CNPGFleetRow[]): StoreRow['health'] return { text: 'Unknown', tone: 'unknown', evidence: 'Its clusters report no archiving result yet' } } -export function CNPGProtection({ data, fleet, namespaces, searchParams, onSetParams, onInspect, inspected, onClearNamespaces }: CNPGScreenProps) { - const clusterFilter = searchParams.get('cluster') +export function CNPGProtection({ + data, + fleet, + namespaces, + searchParams, + onSetParams, + onInspect, + inspected, + onClearNamespaces, + scopeCluster, +}: CNPGScreenProps & { scopeCluster?: { namespace: string; name: string } }) { + const clusterFilter = scopeCluster ? `${scopeCluster.namespace}/${scopeCluster.name}` : searchParams.get('cluster') const rows = useMemo( () => fleet.rows.filter((r) => !clusterFilter || `${r.namespace}/${r.name}` === clusterFilter), [fleet.rows, clusterFilter], @@ -118,24 +128,26 @@ export function CNPGProtection({ data, fleet, namespaces, searchParams, onSetPar [data.objects.scheduledBackups, clusterFilter], ) - const chips = [ + const chips = scopeCluster ? [] : [ ...(clusterFilter ? [{ label: `Cluster: ${clusterFilter}`, onClear: () => onSetParams({ cluster: null }) }] : []), ...namespaceChip(namespaces, onClearNamespaces), ] - const backupsReadable = data.coverage.backups?.state === 'full' || data.coverage.backups?.state === 'partial' + const backupsReadable = data.coverage.backups?.state === 'full' return (
- + {!scopeCluster && ( + + )} @@ -220,7 +232,7 @@ export function CNPGProtection({ data, fleet, namespaces, searchParams, onSetPar rowResource={(b) => cnpgResource('backups', b.metadata?.namespace, b.metadata?.name)} onInspect={onInspect} inspected={inspected} - empty={backupsReadable ? 'No failed backups in the last 7 days.' : 'Backups are not readable with your access.'} + empty={backupsReadable && data.coverage.backups?.state === 'full' ? 'No failed backups in the last 7 days.' : coverageEmpty(data.coverage.backups, 'failed backups')} /> cnpgResource('objectstores', s.namespace, s.name, 'barmancloud.cnpg.io')} onInspect={onInspect} inspected={inspected} - empty={data.coverage.objectStores?.state === 'notInstalled' ? 'The barman-cloud plugin’s ObjectStore kind is not installed.' : 'No ObjectStores in this scope.'} + empty={data.coverage.objectStores?.state === 'notInstalled' ? 'The barman-cloud plugin’s ObjectStore kind is not installed.' : coverageEmpty(data.coverage.objectStores, 'ObjectStores')} footer="ObjectStore has no health status of its own; upload health is inferred from its clusters’ WAL archiving and backup results." /> @@ -261,7 +273,7 @@ export function CNPGProtection({ data, fleet, namespaces, searchParams, onSetPar cell: (s) => ( <> {s.spec?.schedule ?? '—'} - {s.spec?.schedule && {cronToHuman(s.spec.schedule)}} + CNPG cron, seconds first ), }, @@ -290,7 +302,7 @@ export function CNPGProtection({ data, fleet, namespaces, searchParams, onSetPar rowResource={(s) => cnpgResource('scheduledbackups', s.metadata?.namespace, s.metadata?.name)} onInspect={onInspect} inspected={inspected} - empty={data.coverage.scheduledBackups?.state === 'full' ? 'No ScheduledBackups in this scope.' : 'ScheduledBackups are not fully readable with your access.'} + empty={coverageEmpty(data.coverage.scheduledBackups, 'ScheduledBackups')} footer={data.backupsOmitted > 0 ? `${data.backupsOmitted} settled backups older than 7 days are not listed.` : undefined} /> diff --git a/web/src/components/cnpg/CNPGSummaryHost.tsx b/web/src/components/cnpg/CNPGSummaryHost.tsx index bb3dab376..43dcf9217 100644 --- a/web/src/components/cnpg/CNPGSummaryHost.tsx +++ b/web/src/components/cnpg/CNPGSummaryHost.tsx @@ -1,10 +1,8 @@ import type { ReactNode } from 'react' -import { useLocation, useNavigate, useSearchParams } from 'react-router-dom' -import { ArrowLeft } from 'lucide-react' +import { useNavigate } from 'react-router-dom' import { CNPG_BARMAN_OBJECTSTORE_GROUP, CNPG_GROUP, - CNPG_KIND_BY_KEY, CNPGBackupSummary, CNPGClusterSummary, CNPGDatabaseSummary, @@ -23,8 +21,8 @@ import { type NavigateToResource, } from '@skyhook-io/k8s-ui' import { useCNPGFleet } from './useCNPGSidebarWorkspace' -import { cnpgClusterFullPath } from './paths' -import { decodeDrawerTrail, encodeDrawerTrail } from './routes' +import { cnpgClusterFullPath, currentPageLabel } from './paths' +import { useConnection } from '../../context/ConnectionContext' interface SummaryContext { apiKind: string @@ -38,6 +36,7 @@ interface SummaryContext { function ClusterSummaryHost({ namespace, name, context, onNavigate }: SummaryContext) { const navigate = useNavigate() + const { connection } = useConnection() // The workspace is read for the object's own namespace: an explicitly opened // Cluster shows its facts whatever the namespace filter is. const { query, fleet } = useCNPGFleet([namespace]) @@ -58,7 +57,34 @@ function ClusterSummaryHost({ namespace, name, context, onNavigate }: SummaryCon navigate(cnpgClusterFullPath(namespace, name)) }] : undefined} + actions={ + context === 'drawer' + ? [ + { + label: 'Open cluster', + primary: true, + onClick: () => + navigate(cnpgClusterFullPath(namespace, name, connection.context || undefined), { + state: { returnLabel: currentPageLabel(), returnCtx: connection.context }, + }), + }, + { + label: 'Logs', + onClick: () => + navigate(cnpgClusterFullPath(namespace, name, connection.context || undefined, 'logs'), { + state: { returnLabel: currentPageLabel(), returnCtx: connection.context }, + }), + }, + { + label: 'Protection', + onClick: () => + navigate(cnpgClusterFullPath(namespace, name, connection.context || undefined, 'protection'), { + state: { returnLabel: currentPageLabel(), returnCtx: connection.context }, + }), + }, + ] + : undefined + } /> ) } @@ -108,43 +134,10 @@ function renderSummaryFor(ctx: SummaryContext): ReactNode { return Summary ? : null } -const KIND_BY_PLURAL: Record = Object.fromEntries( - Object.values(CNPG_KIND_BY_KEY).map((k) => [k.plural, k.kind]), -) - -// On a workspace screen the drawer URL carries the chain of objects opened -// from inside it; this renders the step back to the previous one. -function DrawerTrailBack({ name, children }: { name: string; children: ReactNode }) { - const location = useLocation() - const [searchParams, setSearchParams] = useSearchParams() - const trail = decodeDrawerTrail(searchParams.get('drawer')) - const current = trail[trail.length - 1] - if (!location.pathname.startsWith('/cnpg') || trail.length < 2 || current?.name !== name) return <>{children} - const prev = trail[trail.length - 2] - const back = () => { - const params = new URLSearchParams(searchParams) - params.set('drawer', encodeDrawerTrail(trail.slice(0, -1))) - setSearchParams(params, { replace: true }) - } - return ( - <> -
- -
- {children} - - ) -} - /** * The composed Overview for CloudNativePG kinds. Returns null for kinds * without one, which keeps the default Overview. */ export function renderCNPGSummary(ctx: SummaryContext): ReactNode { - const node = renderSummaryFor(ctx) - if (!node || ctx.context !== 'drawer') return node - return {node} + return renderSummaryFor(ctx) } diff --git a/web/src/components/cnpg/CNPGView.tsx b/web/src/components/cnpg/CNPGView.tsx index 91b99105b..97a6cce16 100644 --- a/web/src/components/cnpg/CNPGView.tsx +++ b/web/src/components/cnpg/CNPGView.tsx @@ -11,6 +11,7 @@ import { CNPGDeclarations } from './CNPGDeclarations' import { CNPGPooling } from './CNPGPooling' import { CNPGOperator } from './CNPGOperator' import { CNPGScreenGate } from './shared' +import { CNPGDetailPage } from './CNPGDetailPage' import { decodeDrawerTrail, encodeDrawerTrail, parseCNPGRoute, sameResource } from './routes' import { useCNPGFleet, useCNPGSidebarWorkspace } from './useCNPGSidebarWorkspace' @@ -38,7 +39,11 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose const { data: apiResources } = useAPIResources() const { data: counts } = useResourceCounts(namespaces) const { pinned, togglePin, isPinned } = usePinnedKinds() - const sidebarWorkspace = useCNPGSidebarWorkspace({ apiResources, namespaces, active: { screen: route.screen } }) + const sidebarWorkspace = useCNPGSidebarWorkspace({ + apiResources, + namespaces, + active: { screen: route.screen, child: route.detail ? { label: route.detail.name, title: `${route.detail.plural} ${route.detail.namespace}/${route.detail.name}` } : undefined }, + }) const { query, fleet } = useCNPGFleet(namespaces) const drawerParam = searchParams.get('drawer') @@ -68,7 +73,7 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose const next = idx >= 0 ? trail.slice(0, idx + 1) : [...trail, selectedResource] params.set('drawer', encodeDrawerTrail(next)) } - setSearchParams(params, { replace: true }) + setSearchParams(params, { replace: true, state: location.state }) } }, [targetKey, selectedKey]) // eslint-disable-line react-hooks/exhaustive-deps @@ -76,7 +81,7 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose (resource: SelectedResource) => { const params = new URLSearchParams(searchParams) params.set('drawer', encodeDrawerTrail([resource])) - setSearchParams(params, { replace: true }) + setSearchParams(params, { replace: true, state: location.state }) }, [searchParams, setSearchParams], ) @@ -88,7 +93,7 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose if (v === null || v === '') params.delete(k) else params.set(k, v) } - setSearchParams(params, { replace: true }) + setSearchParams(params, { replace: true, state: location.state }) }, [searchParams, setSearchParams], ) @@ -115,6 +120,9 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose categoryWorkspaces={sidebarWorkspace} />
+ {route.detail ? ( + + ) : ( {(data, readyFleet) => { const props = { @@ -141,6 +149,7 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose } }} + )}
) diff --git a/web/src/components/cnpg/paths.ts b/web/src/components/cnpg/paths.ts index e42b6571d..a6f561ae6 100644 --- a/web/src/components/cnpg/paths.ts +++ b/web/src/components/cnpg/paths.ts @@ -1,5 +1,13 @@ -import { buildWorkloadPath } from '../../utils/navigation' +import { cnpgDetailPath } from './routes' -export function cnpgClusterFullPath(namespace: string, name: string): string { - return buildWorkloadPath({ kind: 'clusters', namespace, name, group: 'postgresql.cnpg.io' }) +export function cnpgClusterFullPath(namespace: string, name: string, ctx?: string, tab?: string): string { + return cnpgDetailPath({ plural: 'clusters', namespace, name }, ctx, tab) +} + +/** + * The label for "← back" on the page a push lands on: the title of the page + * being left, which Radar keeps in the document title. + */ +export function currentPageLabel(): string { + return document.title.replace(/\s*·\s*Radar$/, '') || 'previous page' } diff --git a/web/src/components/cnpg/routes.test.ts b/web/src/components/cnpg/routes.test.ts index 15f5a9667..ea390980a 100644 --- a/web/src/components/cnpg/routes.test.ts +++ b/web/src/components/cnpg/routes.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { decodeDrawerTrail, encodeDrawerTrail, parseCNPGRoute, sameResource } from './routes' +import { cnpgDetailKindFor, cnpgDetailPath, decodeDrawerTrail, encodeDrawerTrail, parseCNPGRoute, sameResource } from './routes' describe('CNPG routes', () => { it('parses workspace screens and falls back to Overview for unknown or unavailable ones', () => { @@ -33,4 +33,28 @@ describe('CNPG routes', () => { expect(sameResource(cnpg, capi)).toBe(false) expect(sameResource(cnpg, { ...cnpg })).toBe(true) }) + + it('parses full-detail routes for every CNPG kind and the cluster-scoped placeholder', () => { + expect(parseCNPGRoute('/cnpg/clusters/payments/pg-orders')).toEqual({ + screen: 'overview', + detail: { plural: 'clusters', group: 'postgresql.cnpg.io', namespace: 'payments', name: 'pg-orders' }, + }) + expect(parseCNPGRoute('/cnpg/objectstores/payments/s3-billing').screen).toBe('protection') + expect(parseCNPGRoute('/cnpg/objectstores/payments/s3-billing').detail?.group).toBe('barmancloud.cnpg.io') + expect(parseCNPGRoute('/cnpg/clusterimagecatalogs/_/postgresql-standard').detail?.namespace).toBe('') + expect(parseCNPGRoute('/cnpg/clusters/payments').detail).toBeUndefined() + }) + + it('builds detail paths carrying the context', () => { + expect(cnpgDetailPath({ plural: 'clusters', namespace: 'payments', name: 'pg-orders' }, 'prod', 'logs')).toBe( + '/cnpg/clusters/payments/pg-orders?ctx=prod&tab=logs', + ) + expect(cnpgDetailPath({ plural: 'clusterimagecatalogs', namespace: '', name: 'std' })).toBe('/cnpg/clusterimagecatalogs/_/std') + }) + + it('only claims CNPG kinds from the CNPG groups', () => { + expect(cnpgDetailKindFor('clusters', 'postgresql.cnpg.io')).toBe('clusters') + expect(cnpgDetailKindFor('clusters', 'cluster.x-k8s.io')).toBeNull() + expect(cnpgDetailKindFor('backups', 'velero.io')).toBeNull() + }) }) diff --git a/web/src/components/cnpg/routes.ts b/web/src/components/cnpg/routes.ts index 4621ee9e0..bd3cab25b 100644 --- a/web/src/components/cnpg/routes.ts +++ b/web/src/components/cnpg/routes.ts @@ -10,19 +10,70 @@ export const CNPG_SCREENS: { id: CNPGScreen; label: string; path: string }[] = [ { id: 'operator', label: 'Operator', path: '/cnpg/operator' }, ] +export interface CNPGDetailTarget { + plural: string + group: string + namespace: string + name: string +} + export interface CNPGRoute { screen: CNPGScreen + detail?: CNPGDetailTarget +} + +// The CNPG kinds that have a CNPG-framed full detail, with the workspace +// destination each one lives under. +export const CNPG_DETAIL_KINDS: Record = { + clusters: { group: 'postgresql.cnpg.io', kind: 'Cluster', home: 'overview' }, + backups: { group: 'postgresql.cnpg.io', kind: 'Backup', home: 'protection' }, + scheduledbackups: { group: 'postgresql.cnpg.io', kind: 'ScheduledBackup', home: 'protection' }, + objectstores: { group: 'barmancloud.cnpg.io', kind: 'ObjectStore', home: 'protection' }, + databases: { group: 'postgresql.cnpg.io', kind: 'Database', home: 'declarations' }, + publications: { group: 'postgresql.cnpg.io', kind: 'Publication', home: 'declarations' }, + subscriptions: { group: 'postgresql.cnpg.io', kind: 'Subscription', home: 'declarations' }, + poolers: { group: 'postgresql.cnpg.io', kind: 'Pooler', home: 'pooling' }, + imagecatalogs: { group: 'postgresql.cnpg.io', kind: 'ImageCatalog', home: 'operator' }, + clusterimagecatalogs: { group: 'postgresql.cnpg.io', kind: 'ClusterImageCatalog', home: 'operator', clusterScoped: true }, +} + +export function cnpgDetailKindFor(plural: string, group: string | undefined): string | null { + const p = plural.toLowerCase() + const spec = CNPG_DETAIL_KINDS[p] + return spec && spec.group === (group ?? '') ? p : null } export function parseCNPGRoute(pathname: string): CNPGRoute { - const seg = pathname.replace(/^\/+/, '').split('/') + const seg = pathname.replace(/^\/+/, '').split('/').map((s) => { + try { + return decodeURIComponent(s) + } catch { + return s + } + }) if (seg[0] !== 'cnpg') return { screen: 'overview' } const s = seg[1] ?? '' + const detailSpec = CNPG_DETAIL_KINDS[s] + if (detailSpec && seg[2] && seg[3]) { + return { + screen: detailSpec.home, + detail: { plural: s, group: detailSpec.group, namespace: seg[2] === '_' ? '' : seg[2], name: seg[3] }, + } + } const match = CNPG_SCREENS.find((x) => x.id === s) if (match) return { screen: match.id } return { screen: 'overview' } } +/** Full detail path; `ctx` pins the Kubernetes context the object belongs to. */ +export function cnpgDetailPath(target: Omit, ctx?: string, tab?: string): string { + const params = new URLSearchParams() + if (ctx) params.set('ctx', ctx) + if (tab) params.set('tab', tab) + const qs = params.toString() + return `/cnpg/${target.plural}/${encodeURIComponent(target.namespace || '_')}/${encodeURIComponent(target.name)}${qs ? `?${qs}` : ''}` +} + export function cnpgScreenPath(screen: CNPGScreen): string { return CNPG_SCREENS.find((s) => s.id === screen)?.path ?? '/cnpg' } diff --git a/web/src/components/cnpg/shared.tsx b/web/src/components/cnpg/shared.tsx index 740d96da1..7ce987419 100644 --- a/web/src/components/cnpg/shared.tsx +++ b/web/src/components/cnpg/shared.tsx @@ -6,6 +6,7 @@ import { CNPG_KIND_BY_KEY, PaneLoader, type CNPGFleet, + type CNPGKindCoverage, type CNPGWorkspaceResponse, } from '@skyhook-io/k8s-ui' import type { SelectedResource } from '../../types' @@ -248,6 +249,30 @@ export function SectionTable({ ) } +/** Empty-state text for a collection, derived from how much of it was readable. */ +export function coverageEmpty(cov: CNPGKindCoverage | undefined, noun: string): string { + switch (cov?.state) { + case 'full': + return `No ${noun} in this scope.` + case 'partial': + return `No ${noun} visible. Some namespaces are not readable with your access.` + case 'denied': + return `No access to ${noun}.` + case 'syncing': + return `${noun[0].toUpperCase()}${noun.slice(1)} are still loading.` + case 'error': + return `${noun[0].toUpperCase()}${noun.slice(1)} could not be read.` + default: + return `This kind is not installed.` + } +} + +/** The less complete of two coverages, for collections built from several kinds. */ +export function worstCoverage(...covs: (CNPGKindCoverage | undefined)[]): CNPGKindCoverage | undefined { + const rank: Record = { error: 0, denied: 1, syncing: 2, partial: 3, full: 4, notInstalled: 5 } + return covs.filter(Boolean).sort((a, b) => (rank[a!.state] ?? 9) - (rank[b!.state] ?? 9))[0] +} + export function Mono({ children, title }: { children: ReactNode; title?: string }) { return {children} } diff --git a/web/src/components/resources/ResourceDetailDrawer.tsx b/web/src/components/resources/ResourceDetailDrawer.tsx index d4f778be7..b6e43822b 100644 --- a/web/src/components/resources/ResourceDetailDrawer.tsx +++ b/web/src/components/resources/ResourceDetailDrawer.tsx @@ -1,6 +1,7 @@ import { ResourceDetailDrawer as BaseResourceDetailDrawer } from '@skyhook-io/k8s-ui' import type { SelectedResource } from '../../types' import { WorkloadView } from '../workload/WorkloadView' +import { CNPGDrawerTrailBack } from '../cnpg/CNPGDrawerTrail' interface ResourceDetailDrawerProps { resource: SelectedResource @@ -31,6 +32,9 @@ export function ResourceDetailDrawer(props: ResourceDetailDrawerProps) { return ( {({ resource, expanded, active, initialTab, onClose, onExpand, onExpandIntent, onCancelExpandIntent, onBack, onNavigateToResource, onCollapseToDrawer }) => ( +
+ {!expanded && } +
+
+
)}
) diff --git a/web/src/components/workload/WorkloadView.tsx b/web/src/components/workload/WorkloadView.tsx index 81670bc37..5cde308c9 100644 --- a/web/src/components/workload/WorkloadView.tsx +++ b/web/src/components/workload/WorkloadView.tsx @@ -3,18 +3,20 @@ import { JobRenderer, JobSetRenderer } from '../resources/renderers/JobAdmission import { RayClusterRenderer } from '../resources/renderers/RayClusterRenderer' import { RayServiceRenderer } from '../resources/renderers/RayServiceRenderer' import { KueueWorkloadRenderer } from '../resources/renderers/KueueWorkloadRenderer' -import { useMemo, useEffect, useCallback, useRef, useState } from 'react' +import { useMemo, useEffect, useCallback, useRef, useState, type ReactNode } from 'react' import { useQueries, useQueryClient } from '@tanstack/react-query' -import { useNavigate, useLocation, useSearchParams } from 'react-router-dom' +import { Navigate, useNavigate, useLocation, useSearchParams } from 'react-router-dom' import { workloadPodAwaitsScheduling } from '../capacity/podDemandGate' import { clsx } from 'clsx' import { Terminal, Stethoscope } from 'lucide-react' import { WorkloadView as BaseWorkloadView, + isApiGroup, EditableYamlView, FetchResult, Section, type WorkloadTabType, + type WorkloadExtraTab, type RendererOverrides, type GitOpsOwnerRef, type GitOpsStatus, @@ -148,6 +150,8 @@ import { } from '../resources/renderers/CNPGDeclarativeRenderer' import { CreateResourceDialog } from '../shared/CreateResourceDialog' import { renderCNPGSummary } from '../cnpg/CNPGSummaryHost' +import { CNPGClusterLogs } from '../cnpg/CNPGClusterLogs' +import { cnpgDetailKindFor, cnpgDetailPath } from '../cnpg/routes' import { cleanYamlForDuplicate } from '../../utils/skeleton-yaml' import { useDesktopDownload } from '../../hooks/useDesktopDownload' import { useCompareLauncher } from '../compare/useCompareLauncher' @@ -277,6 +281,11 @@ export function WorkloadViewRoute({ onNavigateToResource }: WorkloadViewRoutePro ) } + const cnpgPlural = cnpgDetailKindFor(kind, group) + if (cnpgPlural) { + return + } + return ( + } + // Workload kinds with stable pod selectors use the aggregated workload logs viewer if (WORKLOAD_LOG_KINDS.has(kind) && (kind !== 'Job' || isCoreBatchJob(apiKind, group))) { return ( From a0cab4a94e8e086fb1875ee84f3244d55a426153 Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Tue, 29 Sep 2026 02:51:05 +0300 Subject: [PATCH 04/17] Gate CNPG activity Event rows on list events and keep context through redirects - Activity rows from Kubernetes Events now also require list events in the namespace; they carry their subject's kind, so the subject check alone let Event messages through. - The /workload redirect to a CNPG detail keeps ctx, pod and every other parameter, and maps a Cluster's timeline tab to Activity. - The drawer trail back link keeps the page's return state. - Activity describes its earliest linked child event as a recorded boundary, not a completeness guarantee. --- internal/server/cnpg_cluster_activity.go | 10 ++++++++++ internal/server/cnpg_cluster_history_test.go | 20 ++++++++++++++++++- .../components/cnpg/CNPGClusterActivity.tsx | 11 +++++----- web/src/components/cnpg/CNPGDrawerTrail.tsx | 2 +- web/src/components/workload/WorkloadView.tsx | 8 +++++++- 5 files changed, 43 insertions(+), 8 deletions(-) diff --git a/internal/server/cnpg_cluster_activity.go b/internal/server/cnpg_cluster_activity.go index b8535287a..59f6fd7fc 100644 --- a/internal/server/cnpg_cluster_activity.go +++ b/internal/server/cnpg_cluster_activity.go @@ -162,7 +162,17 @@ func (s *Server) handleCNPGClusterActivity(w http.ResponseWriter, r *http.Reques } allowed := map[string]bool{} + var eventsAllowed *bool canList := func(e *timeline.TimelineEvent) bool { + if e.Source == timeline.SourceK8sEvent { + if eventsAllowed == nil { + ok := s.canRead(r, "", "events", namespace, "list") + eventsAllowed = &ok + } + if !*eventsAllowed { + return false + } + } key := resourceid.GroupFromAPIVersion(e.APIVersion) + "/" + e.Kind ok, seen := allowed[key] if !seen { diff --git a/internal/server/cnpg_cluster_history_test.go b/internal/server/cnpg_cluster_history_test.go index d781f850e..874526cda 100644 --- a/internal/server/cnpg_cluster_history_test.go +++ b/internal/server/cnpg_cluster_history_test.go @@ -363,9 +363,11 @@ func TestCNPGClusterActivity_DropsKindsTheCallerCannotList(t *testing.T) { name string clusterGet bool backupsListed bool - }{{"reader", true, true}, {"no-backups", true, false}, {"no-clusters", false, true}} { + eventsListed bool + }{{"reader", true, true, true}, {"no-backups", true, false, true}, {"no-clusters", false, true, true}, {"no-events", true, true, false}} { perms := &auth.UserPermissions{AllowedNamespaces: []string{"pgactauth"}} perms.SetCanI("get", cnpgGroup, "clusters", "pgactauth", u.clusterGet) + allow(perms, "", "events", "pgactauth", u.eventsListed) allow(perms, cnpgGroup, "clusters", "pgactauth", true) allow(perms, cnpgGroup, "backups", "pgactauth", u.backupsListed) allow(perms, cnpgGroup, "poolers", "pgactauth", true) @@ -388,6 +390,22 @@ func TestCNPGClusterActivity_DropsKindsTheCallerCannotList(t *testing.T) { t.Errorf("events = %v, want the Cluster and Pod rows", activityIDs(got)) } + sawEvent := false + for _, e := range control.Events { + if e.Source == pkgtimeline.SourceK8sEvent { + sawEvent = true + } + } + if !sawEvent { + t.Fatalf("control: no Kubernetes Event rows: %v", activityIDs(control)) + } + noEvents := decodeActivity(t, env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "no-events", "")) + for _, e := range noEvents.Events { + if e.Source == pkgtimeline.SourceK8sEvent { + t.Errorf("Kubernetes Event row reached a caller who cannot list events: %s", e.ID) + } + } + resp := env.authGet(t, "/api/cnpg/clusters/pgactauth/pg-orders/activity", "no-clusters", "") resp.Body.Close() if resp.StatusCode != http.StatusForbidden { diff --git a/web/src/components/cnpg/CNPGClusterActivity.tsx b/web/src/components/cnpg/CNPGClusterActivity.tsx index 80af1506b..0cceeeb2f 100644 --- a/web/src/components/cnpg/CNPGClusterActivity.tsx +++ b/web/src/components/cnpg/CNPGClusterActivity.tsx @@ -28,11 +28,12 @@ export function CNPGClusterActivity({ namespace, name, onNavigate }: { namespace Kubernetes events and changes for the Cluster, its instances, Backups, Poolers and declarations.
- {q.data?.attributionSince && ( -
- Child-object history is complete since {formatAge(q.data.attributionSince)} ago; earlier Backups and declarations may be missing. -
- )} +
+ {q.data?.attributionSince + ? `The earliest recorded event linking a Backup, Pooler or declaration to this cluster is ${formatAge(q.data.attributionSince)} old. Deleted child objects from before Radar recorded that link are not shown.` + : 'Radar has not recorded any Backup, Pooler or declaration events linked to this cluster yet, so deleted child objects may be missing.'} + {' '}Events for kinds you cannot list are omitted. +
{q.data?.truncated && Showing the most recent events only; narrow the range to see all of them.} {q.error && !q.data ? ( Activity could not be loaded: {q.error instanceof Error ? q.error.message : 'unknown error'} diff --git a/web/src/components/cnpg/CNPGDrawerTrail.tsx b/web/src/components/cnpg/CNPGDrawerTrail.tsx index 58bdf2d51..55d3d9dfe 100644 --- a/web/src/components/cnpg/CNPGDrawerTrail.tsx +++ b/web/src/components/cnpg/CNPGDrawerTrail.tsx @@ -24,7 +24,7 @@ export function CNPGDrawerTrailBack({ resource }: { resource: SelectedResource } const back = () => { const params = new URLSearchParams(searchParams) params.set('drawer', encodeDrawerTrail(trail.slice(0, -1))) - setSearchParams(params, { replace: true }) + setSearchParams(params, { replace: true, state: location.state }) } return (
diff --git a/web/src/components/workload/WorkloadView.tsx b/web/src/components/workload/WorkloadView.tsx index 5cde308c9..98b74f006 100644 --- a/web/src/components/workload/WorkloadView.tsx +++ b/web/src/components/workload/WorkloadView.tsx @@ -283,7 +283,13 @@ export function WorkloadViewRoute({ onNavigateToResource }: WorkloadViewRoutePro const cnpgPlural = cnpgDetailKindFor(kind, group) if (cnpgPlural) { - return + const params = new URLSearchParams(searchParams) + params.delete('apiGroup') + const tab = params.get('tab') + if (cnpgPlural === 'clusters' && (tab === 'timeline' || tab === 'events')) params.set('tab', 'activity') + const base = cnpgDetailPath({ plural: cnpgPlural, namespace, name }) + const qs = params.toString() + return } return ( From 3867b1c060c8819a972b51d4da16a0edcbe374d4 Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Tue, 29 Sep 2026 03:04:01 +0300 Subject: [PATCH 05/17] Fix CNPGView hook dependencies and simplify the summary test's text extraction --- .../k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx | 7 ++++++- web/src/components/cnpg/CNPGView.tsx | 4 ++-- 2 files changed, 8 insertions(+), 3 deletions(-) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx index b5642ceba..b715f2d58 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx @@ -27,7 +27,12 @@ function ws(objects: Partial>, over: Partial]+>/g, '').replace(/'/g, "'").replace(/"/g, '"').replace(/&/g, '&') + return html + .split(/<[^>]*>/) + .join('') + .split(''').join("'") + .split('"').join('"') + .split('&').join('&') } const mainCluster = { diff --git a/web/src/components/cnpg/CNPGView.tsx b/web/src/components/cnpg/CNPGView.tsx index 97a6cce16..0dc899d19 100644 --- a/web/src/components/cnpg/CNPGView.tsx +++ b/web/src/components/cnpg/CNPGView.tsx @@ -83,7 +83,7 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose params.set('drawer', encodeDrawerTrail([resource])) setSearchParams(params, { replace: true, state: location.state }) }, - [searchParams, setSearchParams], + [searchParams, setSearchParams, location.state], ) const setParams = useCallback( @@ -95,7 +95,7 @@ export function CNPGView({ namespaces, selectedResource, onOpenResource, onClose } setSearchParams(params, { replace: true, state: location.state }) }, - [searchParams, setSearchParams], + [searchParams, setSearchParams, location.state], ) const selectKind = useCallback( From d4e5889c794167ca4dbd1cee5c6c734e721aa71e Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Tue, 29 Sep 2026 03:20:57 +0300 Subject: [PATCH 06/17] Address PR review: coverage-aware facts, log stream resume, CNPG log severity - Replication and ObjectStore-sourced facts read "No access" when Pods or ObjectStores are unreadable, instead of asserting absence. - Restore evidence only counts barman-cloud plugin recovery sources. - Inferred ObjectStore health is "Accepting uploads" only when every user cluster reports archiving; the failed-backup window uses completion time. - Log level detection reads CloudNativePG's nested PostgreSQL severity, and live stream lines keep their instance role label. - A restarted instance log stream resumes from the last delivered line instead of replaying its tail; activity Pod rows must match the live Cluster's UID; activity and stream failures are logged. - Workspace components use the shared Tooltip rather than native titles. --- internal/server/cnpg_cluster_activity.go | 21 +++- internal/server/cnpg_cluster_history_test.go | 76 ++++++++++++ internal/server/cnpg_cluster_logs.go | 114 +++++++++++++++++- .../src/components/cnpg/workspace.test.ts | 24 ++++ .../k8s-ui/src/components/cnpg/workspace.ts | 20 ++- .../components/logs/WorkloadLogsViewer.tsx | 1 + .../src/components/logs/useLogBuffer.test.ts | 13 ++ .../src/components/logs/useLogBuffer.ts | 9 ++ web/src/components/cnpg/CNPGDeclarations.tsx | 2 +- web/src/components/cnpg/CNPGDetailPage.tsx | 2 +- web/src/components/cnpg/CNPGOperator.tsx | 12 +- web/src/components/cnpg/CNPGOverview.tsx | 12 +- web/src/components/cnpg/CNPGProtection.tsx | 20 ++- web/src/components/cnpg/shared.tsx | 4 +- 14 files changed, 295 insertions(+), 35 deletions(-) create mode 100644 packages/k8s-ui/src/components/logs/useLogBuffer.test.ts diff --git a/internal/server/cnpg_cluster_activity.go b/internal/server/cnpg_cluster_activity.go index 59f6fd7fc..a3713a455 100644 --- a/internal/server/cnpg_cluster_activity.go +++ b/internal/server/cnpg_cluster_activity.go @@ -1,6 +1,7 @@ package server import ( + "log" "net/http" "sort" "strconv" @@ -71,8 +72,11 @@ func cnpgActivityKindNames() []string { // cnpgRowAttribution decides whether a timeline row is about the named // Cluster. Rows about the Cluster match by identity; instance Pods by their // controller owner; CNPG children by the retained cnpg.io/cluster label, which -// survives their deletion. -func cnpgRowAttribution(e *timeline.TimelineEvent, name string) (matched, labelled bool) { +// survives their deletion. liveUID is the UID of the Cluster that exists now +// under this name, or "" when none does; when set, only Pods it controlled +// count, so a previous same-named Cluster's instances don't merge into a +// recreated one's history. +func cnpgRowAttribution(e *timeline.TimelineEvent, name, liveUID string) (matched, labelled bool) { group := resourceid.GroupFromAPIVersion(e.APIVersion) if _, ok := cnpgActivityKinds[group+"/"+e.Kind]; !ok { return false, false @@ -83,7 +87,8 @@ func cnpgRowAttribution(e *timeline.TimelineEvent, name string) (matched, labell return e.Name == name, false case e.Kind == "Pod" && group == "": o := e.Owner - owned := o != nil && o.Kind == "Cluster" && o.Name == name && resourceid.GroupFromAPIVersion(o.APIVersion) == cnpgGroup + owned := o != nil && o.Kind == "Cluster" && o.Name == name && resourceid.GroupFromAPIVersion(o.APIVersion) == cnpgGroup && + (liveUID == "" || o.UID == liveUID) return owned, owned && labelled default: return labelled, labelled @@ -144,9 +149,16 @@ func (s *Server) handleCNPGClusterActivity(w http.ResponseWriter, r *http.Reques Limit: cnpgActivityScanLimit, }) if err != nil { + log.Printf("[cnpg] Failed to query activity for %s/%s: %v", namespace, name, err) s.writeError(w, http.StatusInternalServerError, err.Error()) return } + var liveUID string + if cache := k8s.GetResourceCache(); cache != nil { + if live, err := findCNPGCluster(r.Context(), cache, namespace, name); err == nil && live != nil { + liveUID = string(live.GetUID()) + } + } scanCapped := len(rows) >= cnpgActivityScanLimit // Attribution is carried by the subject's own rows; K8s Event rows about @@ -155,7 +167,7 @@ func (s *Server) handleCNPGClusterActivity(w http.ResponseWriter, r *http.Reques matched := make([]bool, len(rows)) labelled := make([]bool, len(rows)) for i := range rows { - matched[i], labelled[i] = cnpgRowAttribution(&rows[i], name) + matched[i], labelled[i] = cnpgRowAttribution(&rows[i], name, liveUID) if matched[i] && rows[i].UID != "" { attributedUIDs[rows[i].UID] = true } @@ -227,6 +239,7 @@ func (s *Server) handleCNPGClusterActivity(w http.ResponseWriter, r *http.Reques Limit: 1, }) if err != nil { + log.Printf("[cnpg] Failed to query retention floor for %s/%s: %v", namespace, name, err) s.writeError(w, http.StatusInternalServerError, err.Error()) return } diff --git a/internal/server/cnpg_cluster_history_test.go b/internal/server/cnpg_cluster_history_test.go index 874526cda..6c5f23aee 100644 --- a/internal/server/cnpg_cluster_history_test.go +++ b/internal/server/cnpg_cluster_history_test.go @@ -12,6 +12,7 @@ import ( "time" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" "k8s.io/client-go/kubernetes" "k8s.io/client-go/rest" @@ -447,3 +448,78 @@ func TestCNPGClusterLogsStream_SendsParsedInstanceLines(t *testing.T) { } } } + +func TestCNPGClusterActivity_RecreatedClusterExcludesPreviousIncarnationPods(t *testing.T) { + owner := func(uid string) *timeline.OwnerInfo { + return &timeline.OwnerInfo{Kind: "Cluster", Name: "pg-orders", APIVersion: "postgresql.cnpg.io/v1", UID: uid} + } + // Seeding the cache records the Cluster's own rows, so the Pod rows are + // seeded into a fresh store afterwards and only Pod rows are compared. + podHistory := func(live ...runtime.Object) []string { + k8s.ResetTestDynamicState() + seedCNPGWorkspace(t, cnpgWorkspaceTestKinds, live...) + store := useMemoryTimeline(t) + seedActivity(t, store, "pgrecreate", + activityRow{id: "current-pod", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p-new", age: 10 * time.Minute, owner: owner("orders-uid")}, + activityRow{id: "old-pod", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p-old", age: 2 * time.Hour, owner: owner("previous-uid")}, + activityRow{id: "old-pod-event", apiVersion: "v1", kind: "Pod", name: "pg-orders-1", uid: "p-old", source: timeline.SourceK8sEvent, eventType: timeline.EventTypeWarning, age: 90 * time.Minute}, + ) + resp, err := http.Get(testServer.URL + "/api/cnpg/clusters/pgrecreate/pg-orders/activity") + if err != nil { + t.Fatal(err) + } + var pods []string + for _, e := range decodeActivity(t, resp).Events { + if e.Kind == "Pod" { + pods = append(pods, e.ID) + } + } + return pods + } + + live := withUID(cnpgObj("postgresql.cnpg.io/v1", "Cluster", "pgrecreate", "pg-orders", nil, nil), "orders-uid") + if got := strings.Join(podHistory(live), ","); got != "current-pod" { + t.Fatalf("with the Cluster live: pod events = %s, want only the current incarnation's", got) + } + if got := strings.Join(podHistory(), ","); got != "current-pod,old-pod-event,old-pod" { + t.Fatalf("with the Cluster deleted: pod events = %s, want every incarnation", got) + } +} + +func TestCNPGStreamCursorResumesWithoutReplay(t *testing.T) { + var c cnpgStreamCursor + first := c.restartOptions("postgres", 200, nil) + if first.TailLines == nil || *first.TailLines != 200 || first.SinceTime != nil || !first.Follow || !first.Timestamps || first.Container != "postgres" { + t.Fatalf("first start = %+v", first) + } + line := func(ts, content string) workloadLogEntry { + return workloadLogEntry{Timestamp: ts, Content: content} + } + for _, e := range []workloadLogEntry{line("2026-09-28T14:00:00.1Z", "a"), line("2026-09-28T14:00:05.7Z", "b"), line("2026-09-28T14:00:05.7Z", "c")} { + if !c.admit(e) { + t.Fatalf("fresh line %+v rejected", e) + } + } + + restart := c.restartOptions("postgres", 200, nil) + if restart.TailLines != nil || restart.SinceSeconds != nil || restart.SinceTime == nil || + !restart.SinceTime.Time.Equal(time.Date(2026, 9, 28, 14, 0, 5, 0, time.UTC)) { + t.Fatalf("restart = %+v, want sinceTime at the last delivered second and no tail", restart) + } + + // The resumed follow replays the boundary second. + replayed := []workloadLogEntry{line("2026-09-28T14:00:05.2Z", "earlier in the second"), line("2026-09-28T14:00:05.7Z", "b"), line("2026-09-28T14:00:05.7Z", "c")} + for _, e := range replayed { + if c.admit(e) { + t.Errorf("replayed line %+v admitted", e) + } + } + for _, e := range []workloadLogEntry{line("2026-09-28T14:00:05.7Z", "d"), line("2026-09-28T14:00:06Z", "e")} { + if !c.admit(e) { + t.Errorf("new line %+v rejected", e) + } + } + if c.admit(line("2026-09-28T14:00:05.7Z", "d")) { + t.Error("line before the new last timestamp admitted") + } +} diff --git a/internal/server/cnpg_cluster_logs.go b/internal/server/cnpg_cluster_logs.go index 274a70b3a..19a6436a3 100644 --- a/internal/server/cnpg_cluster_logs.go +++ b/internal/server/cnpg_cluster_logs.go @@ -1,10 +1,12 @@ package server import ( + "bufio" "context" "encoding/json" "errors" "fmt" + "io" "log" "math" "net/http" @@ -15,9 +17,11 @@ import ( "github.com/go-chi/chi/v5" corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" "k8s.io/apimachinery/pkg/labels" "k8s.io/apimachinery/pkg/types" + "k8s.io/client-go/kubernetes" "github.com/skyhook-io/radar/internal/k8s" ) @@ -368,6 +372,7 @@ func (s *Server) handleCNPGClusterLogsStream(w http.ResponseWriter, r *http.Requ } flusher, ok := w.(http.Flusher) if !ok { + log.Printf("[cnpg] Failed to stream logs for %s/%s: response writer does not support flushing", namespace, name) s.writeError(w, http.StatusInternalServerError, "streaming not supported") return } @@ -387,6 +392,7 @@ func (s *Server) handleCNPGClusterLogsStream(w http.ResponseWriter, r *http.Requ logCh := make(chan workloadLogEntry, 1000) var active sync.Map roles := map[string]string{} + cursors := map[string]*cnpgStreamCursor{} start := func(pods []*corev1.Pod) { for _, pod := range pods { roles[pod.Name] = cnpgInstanceRole(pod) @@ -395,12 +401,19 @@ func (s *Server) handleCNPGClusterLogsStream(w http.ResponseWriter, r *http.Requ if _, exists := active.Load(key); exists { continue } + cursor := cursors[key] + if cursor == nil { + cursor = &cnpgStreamCursor{} + cursors[key] = cursor + } + opts := cursor.restartOptions(c, query.tailLines, query.sinceSeconds) streamCtx, streamCancel := context.WithCancel(ctx) - active.Store(key, streamCancel) - go func(podName, containerName, key string) { - defer active.Delete(key) - streamPodLogs(streamCtx, client, namespace, podName, containerName, query.tailLines, query.sinceSeconds, logCh) - }(pod.Name, c, key) + handle := &cnpgStreamHandle{cancel: streamCancel} + active.Store(key, handle) + go func(podName, key string) { + defer active.CompareAndDelete(key, handle) + followCNPGContainerLogs(streamCtx, client, namespace, podName, opts, logCh) + }(pod.Name, key) } } } @@ -417,6 +430,9 @@ func (s *Server) handleCNPGClusterLogsStream(w http.ResponseWriter, r *http.Requ case <-ctx.Done(): return case entry := <-logCh: + if cursor := cursors[entry.Pod+"/"+entry.Container]; cursor != nil && !cursor.admit(entry) { + continue + } if !query.keep(entry) { continue } @@ -452,14 +468,100 @@ func (s *Server) handleCNPGClusterLogsStream(w http.ResponseWriter, r *http.Requ delete(known, podName) active.Range(func(key, value any) bool { if strings.HasPrefix(key.(string), podName+"/") { - value.(context.CancelFunc)() + value.(*cnpgStreamHandle).cancel() active.Delete(key) } return true }) + for key := range cursors { + if strings.HasPrefix(key, podName+"/") { + delete(cursors, key) + } + } sendSSEEvent(w, flusher, "pod_removed", map[string]string{"pod": podName, "reason": "terminated"}) } start(currentPods) } } } + +type cnpgStreamHandle struct { + cancel context.CancelFunc +} + +// cnpgStreamCursor remembers where one container's follow left off, so a +// stream that ends while its Pod is still an instance resumes instead of +// replaying lines the client already has. Only the stream loop touches it. +type cnpgStreamCursor struct { + last time.Time + // atLast holds the contents delivered with timestamp == last. The pod log + // API's sinceTime is second-granular, so a resume replays that second and + // only (timestamp, content) tells a replay from a new line. + atLast map[string]bool +} + +// restartOptions returns the follow request for the next (re)start: the +// caller's window the first time, and from the last delivered second after. +func (c *cnpgStreamCursor) restartOptions(container string, tailLines int64, sinceSeconds *int64) corev1.PodLogOptions { + opts := corev1.PodLogOptions{Container: container, Timestamps: true, Follow: true} + if c.last.IsZero() { + opts.TailLines = &tailLines + opts.SinceSeconds = sinceSeconds + return opts + } + since := metav1.NewTime(c.last.Truncate(time.Second)) + opts.SinceTime = &since + return opts +} + +// admit reports whether an entry is new, recording it when it is. Lines +// arrive in order per container, so anything before the last delivered +// timestamp was already sent. +func (c *cnpgStreamCursor) admit(entry workloadLogEntry) bool { + ts, err := time.Parse(time.RFC3339Nano, entry.Timestamp) + if err != nil { + return true + } + switch { + case ts.Before(c.last): + return false + case ts.Equal(c.last): + if c.atLast[entry.Content] { + return false + } + default: + c.last = ts + c.atLast = map[string]bool{} + } + c.atLast[entry.Content] = true + return true +} + +func followCNPGContainerLogs(ctx context.Context, client kubernetes.Interface, namespace, podName string, opts corev1.PodLogOptions, logCh chan<- workloadLogEntry) { + stream, err := client.CoreV1().Pods(namespace).GetLogs(podName, &opts).Stream(ctx) + if err != nil { + if ctx.Err() == nil { + log.Printf("[cnpg] Failed to follow logs for %s/%s/%s: %v", namespace, podName, opts.Container, err) + } + return + } + defer stream.Close() + reader := bufio.NewReader(stream) + for { + line, err := reader.ReadString('\n') + if line = strings.TrimSuffix(line, "\n"); line != "" && (err == nil || err == io.EOF) { + ts, content := parseLogLine(line) + select { + case logCh <- workloadLogEntry{Pod: podName, Container: opts.Container, Timestamp: ts, Content: content}: + case <-ctx.Done(): + return + } + } + if err != nil { + if err != io.EOF && ctx.Err() == nil { + log.Printf("[cnpg] Failed to read logs for %s/%s/%s: %v", namespace, podName, opts.Container, err) + } + return + } + } +} diff --git a/packages/k8s-ui/src/components/cnpg/workspace.test.ts b/packages/k8s-ui/src/components/cnpg/workspace.test.ts index 75d40b444..f1a6ae1bd 100644 --- a/packages/k8s-ui/src/components/cnpg/workspace.test.ts +++ b/packages/k8s-ui/src/components/cnpg/workspace.test.ts @@ -230,4 +230,28 @@ describe('buildCNPGFleet', () => { const unnamed = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'partial' } } })) expect(unnamed.rows[0].protection.lastSuccessfulBackup.text).toBe('No access to Backups') }) + + it('says no access instead of "no replica pods" when Pods are unreadable', () => { + const fleet = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { pods: { state: 'denied' } } })) + expect(fleet.rows[0].replication.text).toBe('No access to Pods') + }) + + it('does not report the recovery window or last backup as absent when ObjectStores are unreadable', () => { + const c = cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } }) + const p = buildCNPGFleet(resp({ clusters: [c] }, { coverage: { objectStores: { state: 'denied' } } })).rows[0].protection + expect(p.recoveryWindow.text).toBe('No access to ObjectStores') + expect(p.lastSuccessfulBackup.text).toBe('No access to ObjectStores') + }) + + it('ignores recovery sources from other plugins that reuse the barman parameter names', () => { + const src = cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } }) + const other = cluster('pg-x', 'db', { + spec: { + bootstrap: { recovery: { source: 'origin' } }, + externalClusters: [{ name: 'origin', plugin: { name: 'some-other-plugin', parameters: { barmanObjectName: 'store', serverName: 'pg-a' } } }], + }, + }) + const row = buildCNPGFleet(resp({ clusters: [src, other] })).rows.find((r) => r.name === 'pg-a')! + expect(row.protection.restoreValidation.text).toBe('None recorded') + }) }) diff --git a/packages/k8s-ui/src/components/cnpg/workspace.ts b/packages/k8s-ui/src/components/cnpg/workspace.ts index 7276063a8..ce9648905 100644 --- a/packages/k8s-ui/src/components/cnpg/workspace.ts +++ b/packages/k8s-ui/src/components/cnpg/workspace.ts @@ -4,6 +4,7 @@ import type { HealthLevel } from '../resources/resource-utils' import { + CNPG_BARMAN_PLUGIN_NAME, getCNPGClusterBackupConfig, getCNPGClusterBarmanPlugin, getCNPGClusterImageTag, @@ -356,6 +357,7 @@ function lastBackupFact( backups: any[], backupsCov: CNPGKindCoverage, window: ReturnType, + storesUnreadable: boolean, ): CNPGProtectionFacts['lastSuccessfulBackup'] { const ns = cluster.metadata?.namespace const name = cluster.metadata?.name @@ -380,6 +382,7 @@ function lastBackupFact( if (!coverageReadable(backupsCov, ns)) { return { text: coverageUnavailableText(backupsCov, 'Backups'), tone: 'unknown' } } + if (storesUnreadable) return { text: 'No access to ObjectStores', tone: 'unknown' } return { text: 'None observed', tone: 'unknown' } } const best = candidates.reduce((a, b) => (Date.parse(a.at) >= Date.parse(b.at) ? a : b)) @@ -414,7 +417,7 @@ function restoreValidationFact( const sourceName = recovery.source if (sourceName) { const ext = (c.spec?.externalClusters ?? []).find((e: any) => e?.name === sourceName) - const params = ext?.plugin?.parameters + const params = ext?.plugin?.name === CNPG_BARMAN_PLUGIN_NAME ? ext.plugin.parameters : undefined if (store && params?.barmanObjectName === store && (params?.serverName || sourceName) === server) return true } const backupName = recovery.backup?.name @@ -467,10 +470,13 @@ function pgVersion(cluster: any): string | null { return typeof major === 'number' ? String(major) : null } -function replicationFact(cluster: any, pods: CNPGInstance[], hibernated: boolean): CNPGFact { +function replicationFact(cluster: any, pods: CNPGInstance[], hibernated: boolean, podsCov: CNPGKindCoverage): CNPGFact { if (hibernated) return { text: 'Hibernated', tone: 'neutral' } const desired = cluster?.spec?.instances if (desired === 1) return { text: 'Single instance', tone: 'neutral' } + if (!coverageReadable(podsCov, cluster?.metadata?.namespace)) { + return { text: coverageUnavailableText(podsCov, 'Pods'), tone: 'unknown' } + } const replicas = pods.filter((p) => p.role === 'replica') const readyReplicas = replicas.filter((p) => p.ready === true).length if (replicas.length === 0) return { text: 'No replica pods observed', tone: 'unknown' } @@ -616,10 +622,12 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { const desired = typeof cluster?.spec?.instances === 'number' ? cluster.spec.instances : null const window = recoveryWindowFor(cluster, stores) + const storesCov = coverageOf(resp, 'objectStores') + const storesUnreadable = !!getCNPGClusterBarmanPlugin(cluster)?.barmanObjectName && !coverageReadable(storesCov, ns) const protection: CNPGProtectionFacts = { schedule: scheduleFact(cluster, resp.objects.scheduledBackups ?? [], coverageOf(resp, 'scheduledBackups')), destination: destinationFact(cluster), - lastSuccessfulBackup: lastBackupFact(cluster, resp.objects.backups ?? [], coverageOf(resp, 'backups'), window), + lastSuccessfulBackup: lastBackupFact(cluster, resp.objects.backups ?? [], coverageOf(resp, 'backups'), window, storesUnreadable), walArchiving: walFact(cluster), recoveryWindow: window?.from ? { @@ -629,7 +637,9 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { to: window.lastSuccess, source: `ObjectStore ${window.store} status`, } - : { text: 'Not reported', tone: 'unknown' }, + : storesUnreadable + ? { text: coverageUnavailableText(storesCov, 'ObjectStores'), tone: 'unknown' } + : { text: 'Not reported', tone: 'unknown' }, restoreValidation: restoreValidationFact(cluster, clusters, resp.objects.backups ?? []), } const problems = problemsFor(cluster, resp.issues ?? [], resp.audit ?? [], children) @@ -650,7 +660,7 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { hibernated, pgVersion: pgVersion(cluster), catalog: catalogRef(cluster), - replication: replicationFact(cluster, instancePods, hibernated), + replication: replicationFact(cluster, instancePods, hibernated, coverageOf(resp, 'pods')), protection: { ...protection, summary: protectionSummary(protection) }, declarations: declarationsFor(cluster, resp), poolers: poolers diff --git a/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx b/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx index d38b06858..6bbac64e7 100644 --- a/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx +++ b/packages/k8s-ui/src/components/logs/WorkloadLogsViewer.tsx @@ -211,6 +211,7 @@ export function WorkloadLogsViewer({ name, fetchAll, createStream, overrideDownl content: data.content || '', container: data.container || '', pod: data.pod || '', + sourceLabel: data.sourceLabel, podColorIndex: podColorIndexRef.current.get(data.pod || ''), }) } diff --git a/packages/k8s-ui/src/components/logs/useLogBuffer.test.ts b/packages/k8s-ui/src/components/logs/useLogBuffer.test.ts new file mode 100644 index 000000000..22a73bf86 --- /dev/null +++ b/packages/k8s-ui/src/components/logs/useLogBuffer.test.ts @@ -0,0 +1,13 @@ +import { describe, expect, it } from 'vitest' +import { detectLogLevel } from './useLogBuffer' + +describe('detectLogLevel with CloudNativePG records', () => { + it('uses the nested PostgreSQL severity over the instance manager level', () => { + const line = (sev: string) => JSON.stringify({ level: 'info', logger: 'postgres', msg: 'record', record: { error_severity: sev, message: 'x' } }) + expect(detectLogLevel(line('ERROR'))).toBe('error') + expect(detectLogLevel(line('FATAL'))).toBe('error') + expect(detectLogLevel(line('WARNING'))).toBe('warn') + expect(detectLogLevel(line('LOG'))).toBe('info') + expect(detectLogLevel(line('DEBUG1'))).toBe('debug') + }) +}) diff --git a/packages/k8s-ui/src/components/logs/useLogBuffer.ts b/packages/k8s-ui/src/components/logs/useLogBuffer.ts index 3f4f3b93e..0c027cd73 100644 --- a/packages/k8s-ui/src/components/logs/useLogBuffer.ts +++ b/packages/k8s-ui/src/components/logs/useLogBuffer.ts @@ -33,6 +33,15 @@ export function detectLogLevel(content: string): LogLevel { if (trimmed[0] === '{') { try { const obj = JSON.parse(trimmed) + // CloudNativePG wraps each PostgreSQL line in an instance-manager record + // whose own level is usually info; the database's severity is nested. + const pgSeverity = typeof obj.record?.error_severity === 'string' ? obj.record.error_severity.toUpperCase() : '' + if (pgSeverity) { + if (/^(ERROR|FATAL|PANIC)$/.test(pgSeverity)) return 'error' + if (pgSeverity === 'WARNING') return 'warn' + if (/^DEBUG/.test(pgSeverity)) return 'debug' + return 'info' + } const rawLevel = obj.level ?? obj.severity ?? obj.lvl ?? '' // Numeric levels (pino/bunyan): 10=trace, 20=debug, 30=info, 40=warn, 50=error, 60=fatal if (typeof rawLevel === 'number') { diff --git a/web/src/components/cnpg/CNPGDeclarations.tsx b/web/src/components/cnpg/CNPGDeclarations.tsx index fd14248ba..24092ad3b 100644 --- a/web/src/components/cnpg/CNPGDeclarations.tsx +++ b/web/src/components/cnpg/CNPGDeclarations.tsx @@ -239,7 +239,7 @@ export function CNPGDeclarations({ data, fleet, namespaces, searchParams, onSetP {g.cluster} ) : ( - + {g.cluster} )} diff --git a/web/src/components/cnpg/CNPGDetailPage.tsx b/web/src/components/cnpg/CNPGDetailPage.tsx index 41272f77e..9d56bfd52 100644 --- a/web/src/components/cnpg/CNPGDetailPage.tsx +++ b/web/src/components/cnpg/CNPGDetailPage.tsx @@ -130,7 +130,7 @@ export function CNPGDetailPage({
From 9fec7a429b1820c2cd9de8502dadd65e8ef34d00 Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Sat, 3 Oct 2026 03:57:22 +0300 Subject: [PATCH 12/17] Say archiving stopped too when an ObjectStore's backups are failing The failing-backups banner names servers whose cluster also stopped archiving and drops the claim that recovery reaches the newest WAL, and the row keeps its archiving-stopped note while backups fail. --- .../renderers/CNPGObjectStoreRenderer.test.tsx | 17 +++++++++++++++++ .../renderers/CNPGObjectStoreRenderer.tsx | 9 +++++++-- 2 files changed, 24 insertions(+), 2 deletions(-) diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx index 4dee9b9a9..6b73ef95b 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx @@ -97,3 +97,20 @@ describe('a window that has stopped advancing', () => { expect(html).not.toContain('WAL archiving has stopped') }) }) + +describe('ObjectStore failing backups', () => { + const failing = { + ...store, + status: { serverRecoveryWindow: { 'pg-main': { firstRecoverabilityPoint: '2026-08-01T00:00:00Z', lastSuccessfulBackupTime: '2026-08-10T00:00:00Z', lastFailedBackupTime: '2026-08-11T00:00:00Z' } } }, + } + it('does not say recovery still reaches the newest WAL when archiving has stopped too', () => { + const html = renderToString() + expect(html).toContain('WAL archiving has stopped for pg-main') + expect(html).not.toContain('recovery still reaches the newest archived WAL') + expect(html).toContain('WAL archiving has stopped on the cluster behind this server') + }) + it('keeps recovery reaching archived WAL when only base backups fail', () => { + const html = renderToString() + expect(html).toContain('recovery still reaches the newest archived WAL') + }) +}) diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx index e78dab115..c6762f480 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx @@ -44,6 +44,7 @@ export function CNPGObjectStoreRenderer({ const credentialSecret = getCNPGObjectStoreCredentialSecret(data) const cfg = data?.spec?.configuration ?? {} const failing = windows.filter((w) => w.failingSinceLastSuccess) + const archiveStopped = failing.filter((w) => archivingFailing?.has(w.server)).map((w) => w.server) return ( <> @@ -51,7 +52,11 @@ export function CNPGObjectStoreRenderer({ 0 + ? `The most recent base backup failed after the last success, and WAL archiving has stopped for ${archiveStopped.join(', ')}: nothing written since the last archived WAL can be recovered.` + : 'The most recent base backup failed after the last success. While WAL archiving works, recovery still reaches the newest archived WAL, but it replays from an ever older base backup.' + } /> )} @@ -197,7 +202,7 @@ function RecoveryWindowRow({ : 'No backups yet'}
- {stalled && !w.failingSinceLastSuccess && ( + {stalled && (
{/* The timestamps below are real and still describe the last backup that worked. What they no longer describe is a window still growing, and From 545f74e897f76ee21abf1466d2c6e14620dfe53f Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Sat, 3 Oct 2026 04:04:37 +0300 Subject: [PATCH 13/17] Claim recovery from an ObjectStore only after a base backup succeeded With failures and no success the store holds nothing to restore from, and retention isn't shown to move the earliest point. --- .../renderers/CNPGObjectStoreRenderer.test.tsx | 13 ++++++++++--- .../renderers/CNPGObjectStoreRenderer.tsx | 14 +++++++++----- .../components/resources/resource-utils-cnpg.ts | 2 +- 3 files changed, 20 insertions(+), 9 deletions(-) diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx index 6b73ef95b..cbdf9f0fb 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx @@ -106,11 +106,18 @@ describe('ObjectStore failing backups', () => { it('does not say recovery still reaches the newest WAL when archiving has stopped too', () => { const html = renderToString() expect(html).toContain('WAL archiving has stopped for pg-main') - expect(html).not.toContain('recovery still reaches the newest archived WAL') + expect(html).not.toContain('can still replay archived WAL') expect(html).toContain('WAL archiving has stopped on the cluster behind this server') }) - it('keeps recovery reaching archived WAL when only base backups fail', () => { + it('keeps recovery replaying archived WAL when only base backups fail', () => { const html = renderToString() - expect(html).toContain('recovery still reaches the newest archived WAL') + expect(html).toContain('can still replay archived WAL written since') + expect(html).not.toContain('retention') + }) + it('claims no recovery when no base backup ever succeeded', () => { + const never = { ...store, status: { serverRecoveryWindow: { 'pg-main': { lastFailedBackupTime: '2026-08-11T00:00:00Z' } } } } + const html = renderToString() + expect(html).toContain('No base backup has succeeded for pg-main') + expect(html).not.toContain('replay') }) }) diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx index c6762f480..8be2baff3 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx @@ -45,6 +45,7 @@ export function CNPGObjectStoreRenderer({ const cfg = data?.spec?.configuration ?? {} const failing = windows.filter((w) => w.failingSinceLastSuccess) const archiveStopped = failing.filter((w) => archivingFailing?.has(w.server)).map((w) => w.server) + const neverSucceeded = failing.filter((w) => !w.lastSuccessfulBackupTime).map((w) => w.server) return ( <> @@ -53,9 +54,11 @@ export function CNPGObjectStoreRenderer({ variant="error" title={`Backups failing for ${failing.length === 1 ? failing[0].server : `${failing.length} servers`}`} message={ - archiveStopped.length > 0 - ? `The most recent base backup failed after the last success, and WAL archiving has stopped for ${archiveStopped.join(', ')}: nothing written since the last archived WAL can be recovered.` - : 'The most recent base backup failed after the last success. While WAL archiving works, recovery still reaches the newest archived WAL, but it replays from an ever older base backup.' + neverSucceeded.length > 0 + ? `No base backup has succeeded for ${neverSucceeded.join(', ')}, so this store holds nothing to restore from for ${neverSucceeded.length === 1 ? 'it' : 'them'}.` + : archiveStopped.length > 0 + ? `The most recent base backup failed after the last success, and WAL archiving has stopped for ${archiveStopped.join(', ')}: nothing written since the last archived WAL can be recovered.` + : 'The most recent base backup failed after the last success. While WAL archiving works, recovery from that backup can still replay archived WAL written since.' } /> )} @@ -231,8 +234,9 @@ function RecoveryWindowRow({ {w.failingSinceLastSuccess && (
- Every backup since the last success has failed. Recovery replays from that backup, so it takes - longer the longer this lasts, and retention still moves the earliest restorable point forward. + {w.lastSuccessfulBackupTime + ? 'Every backup since the last success has failed. Recovery replays from that backup, so it takes longer the longer this lasts.' + : 'No backup for this server has succeeded, so there is nothing to restore from.'}
)}
diff --git a/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts b/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts index 75bd8fa90..731b37add 100644 --- a/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts +++ b/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts @@ -600,7 +600,7 @@ export interface CNPGObjectStoreRecoveryWindow { firstRecoverabilityPoint?: string lastSuccessfulBackupTime?: string lastFailedBackupTime?: string - /** A failure newer than the last success — the window has stopped advancing. */ + /** A failure newer than the last success, or a failure and no success at all. */ failingSinceLastSuccess: boolean } From c1572dc0c9e7e383f00db2d258df235cfc3a4932 Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Sat, 3 Oct 2026 04:13:06 +0300 Subject: [PATCH 14/17] Read a missing ObjectStore success as not recorded, not as no backup The plugin keeps a successful backup when it fails to update the ObjectStore status, so an absent success time leaves recoverability unestablished rather than proving there is nothing to restore. --- .../renderers/CNPGObjectStoreRenderer.test.tsx | 5 +++-- .../resources/renderers/CNPGObjectStoreRenderer.tsx | 10 +++++----- .../resources/resource-utils-cnpg-objectstore.test.ts | 2 +- .../src/components/resources/resource-utils-cnpg.ts | 2 +- 4 files changed, 10 insertions(+), 9 deletions(-) diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx index cbdf9f0fb..40e78cb40 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.test.tsx @@ -114,10 +114,11 @@ describe('ObjectStore failing backups', () => { expect(html).toContain('can still replay archived WAL written since') expect(html).not.toContain('retention') }) - it('claims no recovery when no base backup ever succeeded', () => { + it('claims nothing about recovery when no successful backup is recorded', () => { const never = { ...store, status: { serverRecoveryWindow: { 'pg-main': { lastFailedBackupTime: '2026-08-11T00:00:00Z' } } } } const html = renderToString() - expect(html).toContain('No base backup has succeeded for pg-main') + expect(html).toContain('No successful base backup is recorded for pg-main') + expect(html).not.toContain('nothing to restore') expect(html).not.toContain('replay') }) }) diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx index 8be2baff3..6799198b4 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx @@ -55,7 +55,7 @@ export function CNPGObjectStoreRenderer({ title={`Backups failing for ${failing.length === 1 ? failing[0].server : `${failing.length} servers`}`} message={ neverSucceeded.length > 0 - ? `No base backup has succeeded for ${neverSucceeded.join(', ')}, so this store holds nothing to restore from for ${neverSucceeded.length === 1 ? 'it' : 'them'}.` + ? `No successful base backup is recorded for ${neverSucceeded.join(', ')}, so recoverability is not established.` : archiveStopped.length > 0 ? `The most recent base backup failed after the last success, and WAL archiving has stopped for ${archiveStopped.join(', ')}: nothing written since the last archived WAL can be recovered.` : 'The most recent base backup failed after the last success. While WAL archiving works, recovery from that backup can still replay archived WAL written since.' @@ -68,7 +68,7 @@ export function CNPGObjectStoreRenderer({ // Configured but empty is NOT the same as healthy. Saying nothing here // would read as "backups are fine" on a store holding nothing.
- No server has reported a backup yet, so there is nothing to restore from this store. + No server has recorded a backup in this store's status, so recoverability is not established.
) : (
@@ -202,7 +202,7 @@ function RecoveryWindowRow({ ? 'Not advancing' : w.lastSuccessfulBackupTime ? 'Recoverable' - : 'No backups yet'} + : 'No backup recorded'}
{stalled && ( @@ -225,7 +225,7 @@ function RecoveryWindowRow({ {w.lastSuccessfulBackupTime ? ( } /> ) : ( - + )} {w.lastFailedBackupTime && ( } /> @@ -236,7 +236,7 @@ function RecoveryWindowRow({
{w.lastSuccessfulBackupTime ? 'Every backup since the last success has failed. Recovery replays from that backup, so it takes longer the longer this lasts.' - : 'No backup for this server has succeeded, so there is nothing to restore from.'} + : 'No successful backup is recorded for this server, so recoverability is not established.'}
)}
diff --git a/packages/k8s-ui/src/components/resources/resource-utils-cnpg-objectstore.test.ts b/packages/k8s-ui/src/components/resources/resource-utils-cnpg-objectstore.test.ts index caa1b7a29..f9a69ed63 100644 --- a/packages/k8s-ui/src/components/resources/resource-utils-cnpg-objectstore.test.ts +++ b/packages/k8s-ui/src/components/resources/resource-utils-cnpg-objectstore.test.ts @@ -77,7 +77,7 @@ describe('getCNPGObjectStoreStatus', () => { it('is unknown, not healthy, when no server has reported', () => { const s = getCNPGObjectStoreStatus(store({})) expect(s.level).toBe('unknown') - expect(s.text).toBe('No backups yet') + expect(s.text).toBe('No backup recorded') }) it('is unhealthy when any server is failing since its last success', () => { diff --git a/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts b/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts index 731b37add..a8edb3a93 100644 --- a/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts +++ b/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts @@ -641,7 +641,7 @@ export function getCNPGObjectStoreRecoveryWindows(resource: any): CNPGObjectStor export function getCNPGObjectStoreStatus(resource: any): StatusBadge { const windows = getCNPGObjectStoreRecoveryWindows(resource) if (windows.length === 0) { - return { text: 'No backups yet', color: healthColors.unknown, level: 'unknown' } + return { text: 'No backup recorded', color: healthColors.unknown, level: 'unknown' } } if (windows.some((w) => w.failingSinceLastSuccess)) { return { text: 'Backups Failing', color: healthColors.unhealthy, level: 'unhealthy' } From a6d98ea3f3283c5365837f9effdab4a4b7afae5e Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Sat, 3 Oct 2026 04:18:40 +0300 Subject: [PATCH 15/17] Word ObjectStore backup outcomes as recorded, not as complete history Status can miss a success whose update failed, so a failure is stated as recorded after the last recorded success, and an entry without timestamps reads as a recovery point not reported. --- .../resources/renderers/CNPGObjectStoreRenderer.tsx | 6 +++--- .../k8s-ui/src/components/resources/resource-utils-cnpg.ts | 2 +- .../src/components/resources/status-overclaim.test.ts | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx index 6799198b4..8b11cd328 100644 --- a/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx +++ b/packages/k8s-ui/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx @@ -57,8 +57,8 @@ export function CNPGObjectStoreRenderer({ neverSucceeded.length > 0 ? `No successful base backup is recorded for ${neverSucceeded.join(', ')}, so recoverability is not established.` : archiveStopped.length > 0 - ? `The most recent base backup failed after the last success, and WAL archiving has stopped for ${archiveStopped.join(', ')}: nothing written since the last archived WAL can be recovered.` - : 'The most recent base backup failed after the last success. While WAL archiving works, recovery from that backup can still replay archived WAL written since.' + ? `A base backup failure is recorded after the last recorded success, and WAL archiving has stopped for ${archiveStopped.join(', ')}: nothing written since the last archived WAL can be recovered.` + : 'A base backup failure is recorded after the last recorded success. While WAL archiving works, recovery from that backup can still replay archived WAL written since.' } /> )} @@ -235,7 +235,7 @@ function RecoveryWindowRow({ {w.failingSinceLastSuccess && (
{w.lastSuccessfulBackupTime - ? 'Every backup since the last success has failed. Recovery replays from that backup, so it takes longer the longer this lasts.' + ? 'A backup failure is recorded after the last recorded success. Restoring from that success replays the WAL archived since, which takes longer the longer this lasts.' : 'No successful backup is recorded for this server, so recoverability is not established.'}
)} diff --git a/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts b/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts index a8edb3a93..ccae941c8 100644 --- a/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts +++ b/packages/k8s-ui/src/components/resources/resource-utils-cnpg.ts @@ -649,7 +649,7 @@ export function getCNPGObjectStoreStatus(resource: any): StatusBadge { // Every timestamp on a recovery window is optional, so a server can be listed // with no recovery point at all. The entry existing is not a backup existing. if (!windows.some((w) => w.firstRecoverabilityPoint || w.lastSuccessfulBackupTime)) { - return { text: 'No recovery point', color: healthColors.unknown, level: 'unknown' } + return { text: 'Recovery point not reported', color: healthColors.unknown, level: 'unknown' } } return { text: 'Recoverable', color: healthColors.healthy, level: 'healthy' } } diff --git a/packages/k8s-ui/src/components/resources/status-overclaim.test.ts b/packages/k8s-ui/src/components/resources/status-overclaim.test.ts index 407b2793d..bef747ccf 100644 --- a/packages/k8s-ui/src/components/resources/status-overclaim.test.ts +++ b/packages/k8s-ui/src/components/resources/status-overclaim.test.ts @@ -141,7 +141,7 @@ describe('getCNPGObjectStoreStatus', () => { // Every timestamp on a RecoveryWindow is optional, so the server can be // listed while holding nothing restorable. expect(getCNPGObjectStoreStatus({ status: { serverRecoveryWindow: { pg: {} } } })) - .toMatchObject({ text: 'No recovery point', level: 'unknown' }) + .toMatchObject({ text: 'Recovery point not reported', level: 'unknown' }) }) it('reports recoverable once a real point exists', () => { From a027865d73f0c6cf07da7349be161b80cd9a79f5 Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Sat, 3 Oct 2026 15:02:08 +0300 Subject: [PATCH 16/17] Keep the CNPG workspace planning doc out of the repository It is internal planning; docs/cnpg.md stops pointing at it. --- docs/cnpg.md | 2 +- docs/plans/CNPG_WORKSPACE.md | 168 ----------------------------------- 2 files changed, 1 insertion(+), 169 deletions(-) delete mode 100644 docs/plans/CNPG_WORKSPACE.md diff --git a/docs/cnpg.md b/docs/cnpg.md index 82c8b02e2..c2d43867f 100644 --- a/docs/cnpg.md +++ b/docs/cnpg.md @@ -59,7 +59,7 @@ Cluster logs (`/api/cnpg/clusters/{ns}/{name}/logs`) need `get pods/log`; Activi ## Not in this version -Runtime data (replication lag, sessions, locks, WAL and slots via the instance manager or Prometheus), Pooler pressure, and operations (Backup now, Switchover, Restart, Hibernate, Restore). See `docs/plans/CNPG_WORKSPACE.md`. +Runtime data (replication lag, sessions, locks, WAL and slots via the instance manager or Prometheus), Pooler pressure, and operations (Backup now, Switchover, Restart, Hibernate, Restore). ## API diff --git a/docs/plans/CNPG_WORKSPACE.md b/docs/plans/CNPG_WORKSPACE.md deleted file mode 100644 index a4727191b..000000000 --- a/docs/plans/CNPG_WORKSPACE.md +++ /dev/null @@ -1,168 +0,0 @@ -# CloudNativePG Workspace — implementation plan - -Source design: Claude Design project `6656e6ed-4e96-41b9-84d2-427d8e9c42fa` (`CNPG Workspace.dc.html`, `CNPG IA Notes.dc.html`). Prototype data is synthetic; this plan maps each screen onto what a real cluster exposes and says where the design must bend. - -Status: DRAFT (revised after Codex cycle 1) — awaiting sign-off. Branch `feature/cnpg-workspace`. Delivered as 6 stacked PRs with a checkpoint after PR3. - ---- - -## 0. Premise check (read first) - -The IA is sound and fits Radar: CNPG is already a first-class integration (10 kinds registered `internal/k8s/dynamic_cache.go:302-313`, renderers for every kind, issues `internal/issues/source_cnpg.go`, audit `pkg/audit/cnpg.go`, demo `scripts/cnpg-demo`). What's missing is a *task-shaped* surface across those kinds. The workspace is the right next step. But five parts of the prototype assume data or mechanisms that don't exist, and building them as drawn would violate the certainty contract: - -| Prototype element | Reality | Plan | -|---|---|---| -| **Runtime · Sessions & locks** (per-pid table with query text, blocking pid) | Instance manager `/pg/status` has no sessions/locks (CNPG `pkg/postgres/status.go`). Per-session rows need `psql` via **pods/exec**. Metrics give only aggregates (`cnpg_backends_total`, `cnpg_backends_waiting_total`, `cnpg_backends_max_tx_duration_seconds`). | v1 ships **aggregates from Prometheus** (sessions by state, waiting-on-locks count, oldest tx age). Per-session table = Needs-input Q1 (exec is a big privilege step). | -| **Runtime denied** message says "denied pods/exec" | Replication/slots/WAL come from `/pg/status` via **pods/proxy** (what `kubectl cnpg status` does: `ProxyGet(scheme,pod,"8000","/pg/status")`). | Denied state names `get pods/proxy` (and, separately, Prometheus unavailability). | -| **Recovery verified** ("Restore drill 6 d ago") | Kubernetes records no restore drills. A Ready Cluster bootstrapped via `bootstrap.recovery` proves a recovery *happened once*, not that today's archive is recoverable. | Column renamed **"Restore evidence"**, **never green**. Default "Not observed" (unknown-grey). When a Cluster in this context has `bootstrap.recovery` sourcing this store/serverName: neutral "Restored into pg-x · created " with that provenance. Needs-input Q2. | -| **URLs carry the Kubernetes context** (`/c/prod-us-east1/cnpg/...`) | Radar's context is server state, not URL (`useSwitchContext` `web/src/api/client.ts:6527`); on switch App keeps pathname **and `location.state`** and drops params (`web/src/App.tsx:1181-1233`). The dangerous case is not a 404 — it's a **same-name object in the new context** silently rendering a different database. | App-wide context-in-URL stays out of scope. CNPG detail routes (and the drawer param) carry **`ctx=`**. If it differs from the active context: render "pg-orders is not in " with **Switch back to ** and **Go to Overview**; never fetch the same-name object. Collections have no `ctx` and simply re-query. `onContextChanged` clears all params today (`web/src/App.tsx:1221-1228`), so it gains a CNPG exception that **preserves `ctx`** on `/cnpg/clusters/*` and `?drawer=` routes (like the existing `openAfter` exception, :1203-1212); a link **without** `ctx` (pasted from outside) is treated as the active context. Tested through the real switch callback. Every write (§2.6) also sends `reviewedContext` + object `uid`. | -| **Return stack** as its own app state | Radar navigation is URL + browser history; no lineage beyond GitOps `?from=` (single level, `GitOpsView.tsx:367-384`). | Implement "← previous task" as **browser history + `location.state.returnLabel`**: every CNPG navigation pushes with a label of the origin; the control reads it and calls `navigate(-1)`. Fresh tab ⇒ no state ⇒ control hidden (IA Q7 for free). Filters/tab/drawer are already in the URL, so Back restores them. No parallel stack. | - -Also note: the Pooling "pressure" and all trend sparklines require Prometheus scraping CNPG's `:9187` / `:9127`. Without series they render "No metrics found" — never zero, and never "PodMonitor not enabled" unless `spec.monitoring.enablePodMonitor` is explicitly false (other scrape setups exist). - -**Premise risk (surfaced, not resolved — Q6):** the strongest journey is "find the unhealthy cluster and understand its availability/protection problem". A smaller design — Overview fleet + Protection feeding the existing detail surfaces — serves it. The heavier parts (Runtime, Actions) carry most of the risk and complexity. The user chose full scope; the phasing below front-loads the fleet journeys and puts an explicit **checkpoint after PR3** before Runtime/Actions. - -## 1. Resolved IA questions (from the notes) - -1. Sidebar highlight on detail opened from kind list → destination-based highlight; return label carries the origin. **Adopt.** -2. Managed role detail home → Cluster **Configuration › Managed roles** anchor. **Adopt.** -3. Which navigations push history → destination changes push; tab/filter/drawer changes `replace`. **Adopt** (matches existing `?full=1` push / collapse replace, `App.tsx:831-836`). -4. ObjectStore ownership → stays in Protection; health always labelled *inferred*, evidence listed per user cluster. **Adopt** (data: `status.serverRecoveryWindow` keyed by serverName + each user Cluster's `ContinuousArchiving` / `LastBackupSucceeded`; the existing host wrapper already derives `clusterForServer`/`archivingFailing`, `web/src/components/resources/renderers/CNPGObjectStoreRenderer.tsx`). -5. Pooling destination vs section → keep both. **Adopt.** -6. Namespace filter vs open object → collections only; an explicitly opened object stays with an "outside the namespace filter" note. **Adopt.** -7. Back on fresh tab → hide control when no return state. **Adopt** (falls out of §0 design). - -Naming: Declarations (not "Databases"), Protection, Pooling, Operator, Overview — adopt the notes' first-use analysis. - -## 2. Architecture - -### 2.1 Routing & nav — `/cnpg/*` view, Resources rail highlighted -- New `ExtendedMainView` `'cnpg'` in `web/src/App.tsx` (`getViewFromPath` :129-148 plus `CRASH_LABELS` :165, `radarPageTitle` :228, `namespaceFilterDisabled` :185 — **enabled** here: collections honour it). -- Routes: `/cnpg` (Overview), `/cnpg/protection`, `/cnpg/declarations`, `/cnpg/pooling`, `/cnpg/operator`, `/cnpg/clusters/:ns/:name/:tab?`. Filters in query (`?filter=attention&cat=&q=&cluster=&drawer=Kind|ns|name`). Router pattern mirrors Capacity (`parseCapacityRoute`, `web/src/components/capacity/shared.tsx:82-106`). -- Primary rail: `PrimaryNavRail` highlights **Resources** for `cnpg` (no new rail item — design invariant). -- Sidebar: `ResourcesSidebar` (k8s-ui) gets an optional **`categoryLinks?: Record`** prop (`{id,label,icon,count?,tone?,active,onSelect}` + optional nested `child`), rendered above a category's kinds under a "Workspace" label, with the kinds collapsed under "Resource kinds". Additive + optional ⇒ no break for Hub. The CloudNativePG category already exists (both groups map to it, `packages/k8s-ui/src/utils/api-resources.ts:242-243`). On `/cnpg/*` the web app renders `ResourcesSidebar` standalone (the package `ResourcesView` supports `hideSidebar`, `packages/k8s-ui/src/components/resources/ResourcesView.tsx:3412`; the web wrapper needs the sidebar's data props lifted into a shared hook so both surfaces feed it identically). -- Workspace problem counts: from the same `/api/cnpg/workspace` query that powers Overview (§2.3) so the sidebar and page never disagree. -- **Navigation model — exactly two mechanisms, never mixed:** - 1. *Browser history* for everything page-level. A `useCNPGNavigate()` wrapper: `push` sets `state.returnLabel` = label of the **current** entry (so the label always describes the entry `navigate(-1)` lands on); `replace` (tab, filter, search, drawer open/close) **copies the existing state forward**. "← {returnLabel}" renders iff `location.state?.returnLabel` exists and calls `navigate(-1)`. Refresh and duplicated tabs keep `history.state`, so the label stays truthful; fresh tabs have none ⇒ hidden. - 2. *Drawer trail* (`?drawer=` list) for in-drawer hops only; its "← Backup x" pops the trail (replace). **Expand** pushes the full page with `returnLabel` = the page under the drawer (the trail is not carried into the page). - - Context switch: App already clears params; CNPG pages also ignore a `returnLabel` whose entry carried a different `ctx`. - - Transition table (Overview→drawer→Backup→ObjectStore→Expand→Back; Resources›Cluster→Expand→Back; fresh-tab deep link; context switch on detail; namespace change with drawer open) is encoded as tests in PR1/PR2. - -### 2.2 Drawer — reuse the global drawer, add a back chain -- Rows open the existing global drawer (`navigateToResource`, `App.tsx:734-752`; shell `packages/k8s-ui/src/components/workload/ResourceDetailDrawer.tsx`). On `/cnpg/*` drawer state is URL-backed via `?drawer=` (same approach as `/resources?resource=`). -- **In-drawer back chain:** add `drawerTrail` to the drawer shell (k8s-ui): links inside the drawer push `{kind,ns,name}` onto the trail and show "← ". Chain lives in the `?drawer=` param as a `~`-joined list of **`Kind.group|ns|name`** refs (group mandatory — CNPG `Cluster`/`Backup` collide with CAPI/KubeBlocks/Velero, which the demo seeds on purpose) so Back/Forward and refresh work. Additive prop. -- **Expand**: for Cluster → navigate to `/cnpg/clusters/ns/name` (full page, §2.4). For other kinds → existing `?full=1` expanded WorkloadView. -- **Focused header** above existing renderers: new k8s-ui components `CNPGSubjectHeader` (problem banner + facts grid + relationship chips + actions) per kind, wired through the existing `rendererOverrides` host wrappers (`web/src/components/workload/WorkloadView.tsx:175-207`) so it appears in drawer, `/resources` drawer and full page alike. Renderers stay unchanged. Contract: the header shows **issues from the Issues engine** (top one + "+N more" → `/issues` filtered to the object) with a next action; the renderer's banners remain the detailed explanation. The header never re-derives health itself (no second precedence rule). - -### 2.3 Data — one gated aggregate endpoint; issues from the Issues engine; display derivation in TS -Why not pure client composition: `/api/resources/{kind}` does **not** per-kind gate namespaced CRDs — `preflightResourceList` only SARs Secrets and LimitRanges ("Other namespaced kinds are deferred", `internal/server/server.go:2139-2172`) and REST callers turn denials into `200 []`. The workspace can't make "No access" vs "none" claims on top of that. And the Issues engine (`internal/issues/source_cnpg.go`, API `pkg/issuesapi/types.go:364`) already detects CNPG problems with concurrent causes preserved; a TS re-derivation would be a divergent second problem system (the parity test only covers phase-string sets, `internal/issues/source_cnpg_parity_test.go`). - -- **`GET /api/cnpg/workspace`** (PR1): for each of the 10 CNPG kinds + instance Pods (`cnpg.io/cluster` label, owner-ref-validated against the Cluster) decide scope per kind **by discovery scope**: namespaced kinds use the capacity pattern (`internal/server/capacity_auth.go:28`: cluster-wide `canRead(list)` else `filterNamespacesByCanRead` over the user's namespaces) intersected with the namespace view filter; **cluster-scoped `ClusterImageCatalog` requires a cluster-scoped `list` SAR, no namespace fallback and no view-filter intersection**; read from the dynamic cache (`listDynamicSynced`), and return `{objects: {kind: [...] (summary-stripped)}, coverage: {kind: full|partial{deniedNamespaces}|denied|notInstalled|syncing}, issues: [CNPG issues for visible objects], audit: [cnpgNoDeclarativeBackup findings], operatorVersion}`. Not-installed ⇒ 200 with `installed:false`. -- **Pure derivation** in k8s-ui `utils/cnpg-workspace.ts`: `buildCNPGFleet(resp) → {rows, counts, categories}`. Unit-tested with fixtures from prototype scenarios + demo shapes. -- **Problems & counts:** "Needs attention" = clusters with ≥1 issue at warning+ attributed to the Cluster or to an object that references it (Backup/ScheduledBackup/Pooler/Database/… via `spec.cluster.name`). Category chips count **affected clusters**, not findings. Row shows top issue + "+N". Neutral **"No declarative backup schedule"** comes from the audit finding `cnpgNoDeclarativeBackup` — worded exactly to its meaning (`pkg/audit/cnpg.go:13`: no ScheduledBackup targets the Cluster; destination/on-demand backups may still exist). "No destination configured" is a separate derived fact from §Protection.2. "Runtime unavailable" category only exists once PR4 lands. -- **Per-cluster facts (only what is observed):** - - Instances: ready/total, primary (`currentPrimary`), designated-primary for replica clusters (`spec.replica.enabled` → "Replica cluster of "). Pills from pod readiness + role label. - - **Replication: "Unknown" without runtime.** Pod readiness and `instancesReportedState` (isPrimary/timeLineID/IP) do not establish streaming. With PR4 runtime: state + lag from `/pg/status`. - - **Protection is four separate facts**, each with its source: - 1. *Schedule*: active / suspended / none (ScheduledBackup `spec.suspend`). - 2. *Destination*: plugin ObjectStore / in-tree `barmanObjectStore` / volumeSnapshot / none (`getCNPGClusterBackupConfig`, `resource-utils-cnpg.ts:457`). - 3. *Last successful backup*: newest of {completed Backup CRs visible, ObjectStore `serverRecoveryWindow[serverName].lastSuccessfulBackupTime`, in-tree `status.lastSuccessfulBackup` when not plugin}, shown with which source won ("from ObjectStore status" / "from Backup pg-x"). Absent everywhere ⇒ "None observed", not "None". - 4. *WAL archiving*: `ContinuousArchiving` condition (absent ⇒ unknown). - Plus recovery window from ObjectStore and "Restore evidence" (§0). - - Declarations: `status.applied` (absent ⇒ Pending), managed roles from `managedRolesStatus.byStatus` and `cannotReconcile`. -- **Other backend additions:** - - `GET /api/cnpg/operator` (PR2) — operator + plugin discovery (Deployments labelled `app.kubernetes.io/name=cloudnative-pg`, plugin Services labelled `cnpg.io/pluginName`; config ConfigMap/Secret **names only**, ConfigMap data only with `get configmaps` in that ns; never Secret values). Each object read gated. - - `GET /api/cnpg/clusters/{ns}/{name}/logs` (+`/stream`) (PR3) — template `internal/server/jobset_logs.go:12`, `collectLogsFromPods(...,bounded=true)`, pods by label **and** controller ownerRef = this Cluster uid, gate `authorizePodLogRead`. Keeps the viewer's `{pod, container, timestamp, content, sourceLabel}` contract; adds optional parsed `level`/`logger` per entry from CNPG's JSON lines (raw passthrough otherwise). - - `GET /api/cnpg/clusters/{ns}/{name}/activity` (PR3) — §2.4. - - `GET /api/cnpg/clusters/{ns}/{name}/status` (PR4) — §2.5. - - `GET .../capabilities` + `POST .../{action}` (PR5) — §2.6. - -### 2.4 Cluster full page — `/cnpg/clusters/:ns/:name/:tab` -New `CNPGClusterPage` on `DetailShell` (`packages/k8s-ui/src/components/shared/DetailShell.tsx`; supply `breadcrumb` **and** `hideBackButton` — breadcrumb alone renders alongside the back nav). Header: return control (§0) · crumb "CloudNativePG / name" · name + `` + status · actions. -Tabs: -- **Overview**: focused problem + "Instances and replication" table + Related resources chips (Poolers, declared DBs, ScheduledBackup, ObjectStore, Pods, catalog, rw/ro/r Services, GitOps source from `argocd.argoproj.io/instance` / Flux labels) + existing `CNPGClusterRenderer` below (with `declared` slot, already wired). -- **Protection**: the three facts, backups table for this cluster, schedule, destination, recovery window; "Backup now" / "Restore to new cluster…". -- **Runtime**: §2.5. -- **Logs**: merged viewer — reuse k8s-ui `WorkloadLogsViewer` (`fetchAll → {pods, logs[{pod, sourceLabel, container, timestamp, content}]}`, `WorkloadLogsViewer.tsx:18`). Needs two **additive** props: `initialPod` and `levelFilter` (reads the optional parsed level). `?pod=` preselects (fleet terminal icon, "Open instance logs"). -- **Activity**: server-side `/api/cnpg/clusters/{ns}/{name}/activity` queries the timeline store (as `handleChanges`, `server.go:3840`, group-aware) for **all CNPG groups in the namespace** plus instance Pods, and keeps events whose object is the Cluster or references it (`spec.cluster.name`, `cnpg.io/cluster` label, owner) — so **deleted** Backups/declarations still appear. Verified gap: `TimelineEvent` stores no `spec.cluster.name` and `ExtractLabels` drops `cnpg.io/cluster` (`pkg/timeline/types.go:107`, `pkg/timeline/converter.go:197`). PR3 therefore **persists the relationship at ingestion**: for `postgresql.cnpg.io` / `barmancloud.cnpg.io` objects, the converter records the referenced cluster (`spec.cluster.name`, or `cnpg.io/cluster` label) as a retained label key — no storage schema change. No name-prefix matching. Events recorded before the upgrade lack it; the tab states "Child-object history is complete since ". Deduped, bounded, one cursor. -- **Configuration**: read-only structured spec: PostgreSQL parameters, storage, bootstrap, managed roles (anchor `#managed-roles`), plugins, monitoring, affinity. Links to YAML. -- **YAML**: existing `EditableYamlView`. -Namespace-filter note when the cluster is outside the current filter. - -### 2.5 Runtime (PR4) — two honest sources, each with its own unavailable state -0. **Security gate before any code (PR4 step 1):** `internal/server/curl.go:126-132` asserts apiserver `services/proxy` forwards the caller's credentials to the workload. Upstream apiserver deletes `Authorization` after successful authentication and strips `Impersonate-*`, so I expect pods/proxy to forward neither — but PR4 **proves it** with a demo-cluster echo pod recording received headers under both kubeconfig-token and in-cluster SA auth. If anything credential-bearing arrives, pods/proxy is dropped and runtime is Prometheus-only. -1. **Instance manager** via impersonated `pods/proxy` GET to `https::8000/pg/status` (scheme per pod: `--status-port-tls` in container command, as `remote.GetStatusSchemeFromPod`). Client = `getClientForRequest(r)` (`server.go:5227`) so apiserver enforces caller RBAC; pre-gate with `canReadSubresource(r,"","pods","proxy",ns,"get")` for a capability flag. Denial detection mirrors `pkg/probe/probe.go:720`. Yields: per-instance role, LSNs, WAL receiver, `replicationInfo[]` lags (`writeLag/flushLag/replayLag`, `syncState`), `replicationSlotsInfo[]`, archiver (`lastArchivedWAL[Time]`, `lastFailedWAL[Time]`, `readyWalFiles`), `pendingRestart`. Fan-out bounded (≤ instances, 5s timeout each), per-instance error preserved (one unreachable replica ≠ whole tab down). Response strips the embedded `pod` object. -2. **Prometheus** (optional, via `prometheuspkg.GetClient()`; gated with `canRead(clusters get)` in ns). **Isolation contract:** every query is built server-side (no user PromQL), selects `namespace=""` **and** `pod=~""` (regex-escaped), and applies whatever cluster-identity label Radar's curated resource charts already use for shared multi-cluster Prometheus (reuse that helper; don't invent one); duplicate scrapes collapsed with `max by (pod, …)`. Series: sessions by state (`cnpg_backends_total`), waiting (`cnpg_backends_waiting_total`), oldest tx (`cnpg_backends_max_tx_duration_seconds`), xid age, DB sizes (`cnpg_pg_database_size_bytes`), WAL/archiver rates, 1h trends (`QueryRange`), pooler `cnpg_pgbouncer_pools_cl_waiting/sv_active` — **pooler pods are selected separately**: Pooler → its generated Deployment (`resource-utils-cnpg.ts:1004`) → owned ReplicaSets → Pods, each hop from the cache with ownerRef validation, gated on `get poolers` in ns, same isolation rules. Absent Prometheus or series ⇒ "No metrics (Prometheus not connected / PodMonitor not enabled)". -Sub-tabs: Replication · Sessions · Transactions · Storage & WAL · Slots · Trends. Every value carries its source and sample time ("from instance manager · 5 s ago", "Prometheus · 30 s"). Three distinct states per source, never merged: **denied** (SAR false or apiserver Forbidden on proxy → names the grant), **unreachable** (permission OK, pod didn't answer — per instance), **absent** (no Prometheus / no series). Proxy denial with working metrics shows metrics; the design's dashed "Runtime data unavailable" card appears only when both sources are unavailable, listing what's still available. Fleet "Runtime unavailable" category = runtime capability false. - -### 2.6 Actions (PR5) — impersonated, SAR-gated, confirmed -Capabilities endpoint pattern from `handleRolloutCapabilities` (`internal/server/rollouts_handlers.go:106`), operations from `handleRolloutOperation` (:50) with `auth.AuditLog`, `getDynamicClientForRequest` (nil ⇒ 503, never SA fallback), error mapping like `writeRolloutError` (:224). Every POST carries `reviewedContext` + Cluster `uid` + `resourceVersion` seen in the confirm dialog; server rejects (409) on context/uid mismatch and **re-validates preconditions server-side** (not terminating, not hibernated, no switchover in flight, target is a current ready non-fenced instance of *this* Cluster). UI distinguishes *requested* (patch accepted) from *completed* (observed in status: `currentPrimary == target`, hibernation condition, Backup phase). - -| Action | Mechanism (as `kubectl cnpg` does) | SAR | -|---|---|---| -| Backup now | create `Backup` `{cluster, method}` where method ∈ the cluster's **configured** methods (plugin → `pluginConfiguration.name: barman-cloud.cloudnative-pg.io`; `barmanObjectStore`; `volumeSnapshot`) — dialog picks when >1; `target` default; name `-`; label `cnpg.io/cluster` | `create backups` | -| Switchover | status merge-patch `targetPrimary`, `targetPrimaryTimestamp`, `phase="Switchover in progress"`, `phaseReason` **with `metadata.resourceVersion` in the patch** (optimistic lock, as `kubectl cnpg promote`). On conflict: re-read, re-validate preconditions, and return 409 "cluster changed since you confirmed" — **no blind retry**. (Rollouts' `patchStatusThenSpec`, `pkg/rollouts/rollouts.go:220`, has no lock — shape reference only.) | `patch clusters/status` (separately from `patch clusters`) | -| Restart | annotation `kubectl.kubernetes.io/restartedAt` | `patch clusters` | -| Reload config | annotation `cnpg.io/reloadedAt` | `patch clusters` | -| Hibernate / Rehydrate | annotation `cnpg.io/hibernation=on/off`; progress from condition `cnpg.io/hibernation` | `patch clusters` | -| Restore to new cluster | **no bespoke endpoint**: dialog builds a new Cluster manifest — plugin: `bootstrap.recovery.source: origin` + `externalClusters: [{name: origin, plugin: {name: barman-cloud.cloudnative-pg.io, parameters: {barmanObjectName, serverName}}}]`; in-tree: `externalClusters[].barmanObjectStore` copied from source; or `recovery.backup.name` (same ns). Same image/catalog major as source, storage copied, optional `recoveryTarget.targetTime` bounded by the observed recovery window. Opens in the existing review flow: `/api/resources/preview` then `/api/resources/apply?mode=create&reviewedContext=` (strict create — a name collision fails, never updates, `server.go:4131-4191`). Dialog states that dry-run validates admission only, not archive reachability; new cluster has no WAL archiving unless added (warned, pointing at a different serverName). | apiserver-enforced | -| Edit YAML | existing editor | existing | - -Confirm dialogs name the effect ("Switchover promotes pg-orders-2; pg-orders-1 restarts as replica; writes pause for seconds"). Switchover target list excludes non-ready/fenced instances; disabled with reason when the cluster is mid-switchover/hibernated/terminating. Mutation toasts via React Query `meta` (CLAUDE.md frontend rule). - -## 3. Phased delivery (each PR independently shippable) - -| PR | Scope | Backend | Size | -|---|---|---|---| -| **1 Foundation + Overview** | `/cnpg` view + routes; sidebar `categoryLinks` (k8s-ui); `useCNPGWorkspace` + `buildCNPGFleet` (+ tests); Overview fleet (segments Needs attention/All, category chips, search, ns chip, issue column, instance pills; logs icon hidden until PR3); Cluster focused header in drawer; empty / not-installed / partial-coverage states; `useCNPGNavigate` + transition tests; `ctx` guard | `/api/cnpg/workspace` | L | -| **2 Protection · Declarations · Pooling · Operator + supporting headers** | four screens; composed Overview summaries (`CNPGSummary`) + Overview · Spec & status · YAML tabs for all supporting kinds, drawer and `/cnpg/:kind/...` full detail (§7); drawer back chain | `/api/cnpg/operator` | L | -| **3 Cluster full page** | tabs Overview (with K8s-only replication topology), Protection, Logs, Activity, Configuration, YAML; Expand → page; Resources-nav flyout on detail < 1600 px; not-in-context state; ns-filter note | merged logs (+stream) | L | -| — | **Checkpoint with user** after PR3: is Runtime/Actions still wanted as designed? (Q6) | | | -| **4 Runtime** | step 1 proxy-header proof (§2.5.0); demo gains Prometheus + PodMonitor fixture; Runtime tab; fleet lag/runtime category; pooler pressure | `/status` (pods/proxy) + Prometheus queries | M-L | -| **5 Actions** | demo gains a MinIO-backed `live-backup` mode (successful backup + restore; README constraints respected); capabilities + actions + Restore dialog via preview/apply | capabilities + action endpoints | M | -| **6 Docs + polish** | `docs/cnpg.md` (certainty contract like `docs/capacity.md`), `docs/integrations.md`, CLAUDE.md reference-docs row + endpoints list, README row fix (`README.md:549` stale). Docs for each endpoint land with its PR; this PR consolidates. | — | S | - -MCP: no new tools in this effort (the workspace is UI composition over data MCP already exposes). Revisit after. - -## 4. Testing -- k8s-ui unit: `buildCNPGFleet` fixtures (healthy, WAL failing, backup failed, declaration failed, forbidden kind, no backups configured, runtime unknown); sidebar `categoryLinks` render; header per kind. -- Go: handler tests using `newAuthTestServer`/`dynamicfake` patterns (`internal/server/server_auth_test.go:289`, `velero_handlers_test.go:328`): logs gate 403, status proxy denial → `runtime.denied`, capability SAR matrix (patch clusters vs clusters/status), action error mapping, no SA fallback when impersonation nil. Keep `cnpg_handlers_test.go` source-grep contracts intact. -- Coverage matrix tests for `/api/cnpg/workspace`: per-kind cluster-wide vs namespaced vs denied, colliding groups (Velero Backup, KubeBlocks Cluster from the demo), not installed. -- Context tests: same-name Cluster in two contexts → detail shows "not in this context", no fetch; write with stale `reviewedContext`/uid → 409. -- `/visual-test` on `make cnpg-demo` per PR with UI (frozen for states; `live` + new Prometheus / MinIO modes for PR4/5). -- `make tsc`, `make test`. - -## 5. Risks -- **Sidebar prop in k8s-ui** is public surface for Hub — additive only; Hub unaffected until it opts in. -- **Fleet payload size** on large fleets (design's 48-cluster variant): `/api/cnpg/workspace` returns summary-stripped objects; Backups are the long tail — cap to last 7 days + newest completed per cluster, with a count of omitted. -- **pods/proxy forwards caller credentials to the pod** (`internal/server/curl.go:126-132` caveat). Instance manager `/pg/status` is unauthenticated and read-only, so acceptable; we only ever GET `/pg/status`. -- **CNPG version drift**: phases matched on English sentences (demo README warning); `/pg/status` field names verified at v1.27 and v1.30. -- **Issues attribution** of child-object issues to clusters depends on `spec.cluster.name`; ObjectStore issues map to clusters via plugin `barmanObjectName`/`serverName` (same logic the ObjectStore host wrapper uses). - -## 6. Needs-input (blocking sign-off) -- **Q1 Sessions table**: ship aggregates only (Prometheus) — recommended — or add per-session rows via `pods/exec psql` (needs `create pods/exec`, shows query text = potential PII)? -- **Q2 Restore evidence**: keep the column as "Restore evidence" (never green; neutral provenance when a restored cluster exists) — recommended — or drop it? -- **Q3 Write actions scope** (RBAC posture): all of Backup now, Switchover, Restart, Reload, Hibernate/Rehydrate, Restore-via-apply — or a subset for v1? Should they be hidden behind an existing read-only/`--no-actions` setting if Radar has one? -- **Q4 Context guard**: `ctx=` only on CNPG detail/drawer URLs (recommended) vs app-wide context-in-URL as a separate project first? -- **Q5 PR cadence**: 6 stacked PRs off `feature/cnpg-workspace` (recommended) vs one long-lived branch merged at the end. -- **Q6 Premise checkpoint**: agree to pause after PR3 (fleet + Protection + Declarations + Pooling + Operator + Cluster page with Logs/Activity) to decide on Runtime/Actions with real usage in hand? - -## 7. Design revision 2 (Sep 29) — deltas folded in - -Revision 2 responds to `uploads/PROTOTYPE-REVIEW.md`; the design project also now carries `uploads/PLAN.md` (the IA decision doc) and `uploads/DESIGN-PROMPT.md`. The v1 file is kept as `CNPG Workspace v1.dc.html`. - -| Rev 2 change | Effect on this plan | -|---|---| -| **One composed detail**: CNPG drawers/pages have tabs **Overview · Spec & status · YAML**. Overview = composed fact hierarchy (e.g. Backup: Outcome / Relationships; ObjectStore: Upload health · inferred / Recovery window / Destination / Used by; Database: Declared / Reconciled / Source and target). The existing renderer moves to **Spec & status**, nothing repeats. | Replaces §2.2's "focused header above the existing renderer". Each CNPG kind gets a `CNPGSummary` (k8s-ui, pure) for Overview; existing renderers unchanged under Spec & status. Resolves Codex #13 duplication. | -| **Every CNPG kind has a CNPG full detail** (`uploads/PLAN.md`: the 10 kinds are "CNPG detail destinations"; expanding from Resources enters it, return goes back to Resources). | §2.2 "Expand for other kinds → `?full=1`" is dropped: route `/cnpg/:kind/:ns/:name/:tab` (`_` ns for ClusterImageCatalog) for all 10 kinds; Cluster keeps its 7 tabs. Resources drawer Expand on a CNPG kind navigates there with `returnLabel`. | -| **Replication topology** (`CNPGReplication`): primary card → replica cards with LSN, timeline, sync mode, lag pill + 60 s lag bar, Pooler chips in front; logs/pod links per instance. With runtime denied: "Topology from Kubernetes status. Lag and LSN need runtime access." | Matches §2.3 (replication Unknown without runtime). Topology from K8s (roles), LSN/lag/sync from `/pg/status` (PR4). Component lands in PR3 in its K8s-only form. | -| **Trends**: y-axis scale, threshold line, 15 min / 1 h range, hover values, hatched **gaps with reason** (not zero), per-chart source + sample coverage; **click a bar → that instance's logs ±6 min** as a removable chip. "Latest log lines" under replication. | PR4: Prometheus `QueryRange`; gaps = missing samples. Logs endpoint needs `sinceTime` + client-side upper bound for the ±6 min window (K8s logs API has no until). | -| **Scoped counts**: sidebar badges and kind counts follow the namespace filter with a "Counts for namespace …" note; Protection count = clusters with failing or unconfigured protection everywhere. | Already one source (`/api/cnpg/workspace`). Kind inventory counts come from `/resource-counts` which is already namespace-scoped. "Unconfigured" must use our wording ("no declarative schedule" / "no destination"), not "No backups". | -| **Qualified claims**: "Restore validation: None recorded"; ObjectStore "Uploads failing (inferred)" with source + time; recovery window "per Cluster status at 14:19"; controller phase labelled "reported by CNPG; Radar findings are separate"; ObjectStore Spec tab explains why "Recoverable" coexists with failing uploads. | Adopt the wording. **Two divergences kept:** (a) the prototype still shows one *green* "Restore drill 6 d ago" linking to Activity — Kubernetes records no drills, so we stay never-green (Q2); (b) recovery window "per Cluster status" — for plugin clusters `status.firstRecoverabilityPoint` is deprecated/unset, so we source it from ObjectStore `serverRecoveryWindow` and cite that. | -| **Correlation hints**: ObjectStore problem cites "Secret s3-billing-creds changed 2 minutes earlier"; Database problem explains "owner not among managed roles". | Both derivable (timeline event on the referenced credentials Secret; `spec.owner` vs `spec.managed.roles`). Shown as adjacent facts, not asserted causes ("Secret changed at 09:03" — no "because"). Roles may exist outside `managed.roles`, so the Database hint only appears when the operator error names the role. | -| **Return = drilldown only**: sidebar and global-nav hops clear the return stack. | Matches §2.1: only drilldown pushes set `returnLabel`; sidebar hops push without it ⇒ control hidden. Add to transition tests. | -| **Row actions**: labelled "Logs" (file icon, not terminal) + "Open →"; row inspects; no instruction sentence. | PR1 (Logs button appears once PR3 lands). | -| **Layout**: icon rail below 1600 px; on detail screens below 1600 px the Resources nav moves into a ☰ flyout beside the crumb; related resources stack under evidence below 1080 px main column; Runtime sub-sections are one segmented row. | Radar's rail already has an unpinned icon mode (`web/src/components/nav/PrimaryNavRail.tsx:48`). New work: sidebar flyout on CNPG detail routes (PR3). | -| **Kinds collapsed on workspace/detail**, open on Resources lists; user toggle wins. | `categoryLinks` prop gains `defaultKindsCollapsed`. | - -Still open from the design side: its notes say DESIGN.md / PLAN.md / the review "have not been attached", but they are now in `uploads/` — rev 2 may predate reading them, so a DESIGN.md conformance pass remains our job during implementation. From e8605149e037b935baf916fce8740d99a073dc75 Mon Sep 17 00:00:00 2001 From: Nadav Erell Date: Sun, 4 Oct 2026 12:19:38 +0300 Subject: [PATCH 17/17] CloudNativePG: Logs tab for Clusters without related Pods; one store-health inference A CNPG Cluster's instance Pods are its children, not related Pods, so the Logs tab no longer depends on them. Protection's Destinations column now uses the same inference as the ObjectStore summary, so recovery windows left by clusters that no longer use the store don't read as failures. --- packages/k8s-ui/src/components/cnpg/index.ts | 6 +++ .../src/components/workload/WorkloadView.tsx | 6 +++ .../workload-logs-availability.test.ts | 6 +++ web/src/components/cnpg/CNPGProtection.tsx | 40 ++++++++++--------- 4 files changed, 39 insertions(+), 19 deletions(-) diff --git a/packages/k8s-ui/src/components/cnpg/index.ts b/packages/k8s-ui/src/components/cnpg/index.ts index 026c0b142..08bd46e1f 100644 --- a/packages/k8s-ui/src/components/cnpg/index.ts +++ b/packages/k8s-ui/src/components/cnpg/index.ts @@ -6,3 +6,9 @@ export * from './CNPGObjectStoreSummary' export * from './CNPGDeclarativeSummary' export * from './CNPGPoolerSummary' export * from './CNPGImageCatalogSummary' +export { + inferredObjectStoreHealth, + usersOfObjectStore, + type CNPGObjectStoreHealth, + type CNPGObjectStoreUser, +} from './relations' diff --git a/packages/k8s-ui/src/components/workload/WorkloadView.tsx b/packages/k8s-ui/src/components/workload/WorkloadView.tsx index 0e0805d8d..c6e1f3c93 100644 --- a/packages/k8s-ui/src/components/workload/WorkloadView.tsx +++ b/packages/k8s-ui/src/components/workload/WorkloadView.tsx @@ -1917,6 +1917,7 @@ const LOGS_TAB_WITHOUT_PODS_KINDS = new Set([ 'clusterworkflowtemplates', 'scaledjobs', 'jobsets', + 'clusters', ]) const RUNTIME_WORKLOAD_OVERVIEW_KINDS = new Set(['deployments', 'statefulsets', 'daemonsets', 'jobs', 'cronjobs']) const ROLLOUT_STATUS_KINDS = new Set(['deployments', 'statefulsets', 'daemonsets', 'rollouts']) @@ -1933,6 +1934,11 @@ export function supportsLogsWithoutPods( if (normalizedKind === 'jobsets') { return group === 'jobset.x-k8s.io' && apiVersion === 'jobset.x-k8s.io/v1alpha2' } + // A CloudNativePG Cluster's instance Pods are its children, not related + // Pods, and its logs are resolved server-side from the Cluster itself. + if (normalizedKind === 'clusters') { + return group === 'postgresql.cnpg.io' || !!apiVersion?.startsWith('postgresql.cnpg.io/') + } return true } diff --git a/packages/k8s-ui/src/components/workload/workload-logs-availability.test.ts b/packages/k8s-ui/src/components/workload/workload-logs-availability.test.ts index c78a3fd60..c34b02a02 100644 --- a/packages/k8s-ui/src/components/workload/workload-logs-availability.test.ts +++ b/packages/k8s-ui/src/components/workload/workload-logs-availability.test.ts @@ -17,4 +17,10 @@ describe('supportsLogsWithoutPods', () => { expect(supportsLogsWithoutPods('jobs', 'Job', 'batch', 'batch/v1')).toBe(true) expect(supportsLogsWithoutPods('jobs', 'Job', 'example.io', 'example.io/v1')).toBe(false) }) + + it('offers a CloudNativePG Cluster its merged instance logs, but not other Cluster kinds', () => { + expect(supportsLogsWithoutPods('clusters', 'Cluster', 'postgresql.cnpg.io', 'postgresql.cnpg.io/v1')).toBe(true) + expect(supportsLogsWithoutPods('clusters', 'Cluster', undefined, 'postgresql.cnpg.io/v1')).toBe(true) + expect(supportsLogsWithoutPods('clusters', 'Cluster', 'cluster.x-k8s.io', 'cluster.x-k8s.io/v1beta1')).toBe(false) + }) }) diff --git a/web/src/components/cnpg/CNPGProtection.tsx b/web/src/components/cnpg/CNPGProtection.tsx index 2316582df..7bbbc0c9e 100644 --- a/web/src/components/cnpg/CNPGProtection.tsx +++ b/web/src/components/cnpg/CNPGProtection.tsx @@ -7,10 +7,11 @@ import { getCNPGBackupStatus, getCNPGClusterBarmanPlugin, getCNPGObjectStoreDestination, - getCNPGObjectStoreRecoveryWindows, getCNPGScheduledBackupStatus, + inferredObjectStoreHealth, isApiGroup, toneTextClass, + usersOfObjectStore, type CNPGFleetRow, type HealthLevel, } from '@skyhook-io/k8s-ui' @@ -61,29 +62,30 @@ interface StoreRow { health: { text: string; tone: HealthLevel; evidence: string } } -function inferStoreHealth(store: any, users: CNPGFleetRow[]): StoreRow['health'] { - if (users.length === 0) { - return { text: 'Unknown', tone: 'unknown', evidence: 'No visible cluster uses this store' } - } - const failingArchiving = users.filter((u) => u.protection.walArchiving.tone === 'unhealthy') - const windows = getCNPGObjectStoreRecoveryWindows(store) - const failingBackups = windows.filter((w) => w.failingSinceLastSuccess) - if (failingArchiving.length > 0 || failingBackups.length > 0) { +// The same inference the ObjectStore summary shows, so the two never disagree: +// only the recovery windows of clusters that use the store now count. +function storeHealth(store: any, users: CNPGFleetRow[]): StoreRow['health'] { + const { summary, evidence } = inferredObjectStoreHealth(store, usersOfObjectStore(store, users.map((u) => u.cluster))) + const names = (list: typeof evidence) => list.map((e) => e.cluster.name).join(', ') + if (evidence.length === 0) return { text: 'Unknown', tone: 'unknown', evidence: summary.text } + if (summary.tone === 'unhealthy') { + const archiving = evidence.filter((e) => e.archiving.tone === 'unhealthy') + const backups = evidence.filter((e) => e.window?.failingSinceLastSuccess) const parts = [ - failingArchiving.length > 0 ? `WAL archiving failing on ${failingArchiving.map((u) => u.name).join(', ')}` : null, - failingBackups.length > 0 ? `a backup failed after the last success for ${failingBackups.map((w) => w.server).join(', ')}` : null, + archiving.length > 0 ? `WAL archiving failing on ${names(archiving)}` : null, + backups.length > 0 ? `a backup failed after the last success for ${names(backups)}` : null, ].filter(Boolean) - return { text: 'Uploads failing', tone: 'unhealthy', evidence: `Inferred: ${parts.join('; ')}` } + return { text: summary.text, tone: summary.tone, evidence: `Inferred: ${parts.join('; ')}` } } - const archiving = users.filter((u) => u.protection.walArchiving.tone === 'healthy') - if (archiving.length === users.length) { - return { text: 'Accepting uploads', tone: 'healthy', evidence: `Inferred from WAL archiving on ${archiving.map((u) => u.name).join(', ')}` } + const archiving = evidence.filter((e) => e.archiving.tone === 'healthy') + if (summary.tone === 'healthy') { + return { text: summary.text, tone: summary.tone, evidence: `Inferred from WAL archiving on ${names(archiving)}` } } return { - text: 'No failures reported', - tone: 'unknown', + text: summary.text, + tone: summary.tone, evidence: archiving.length > 0 - ? `Archiving on ${archiving.map((u) => u.name).join(', ')}; no archiving result from the others` + ? `Archiving on ${names(archiving)}; no archiving result from the others` : 'Its clusters report no archiving result yet', } } @@ -126,7 +128,7 @@ export function CNPGProtection({ const users = fleet.rows.filter( (r) => r.namespace === ns && getCNPGClusterBarmanPlugin(r.cluster)?.barmanObjectName === name, ) - return { key: `${ns}/${name}`, namespace: ns, name, destination: getCNPGObjectStoreDestination(s), users, health: inferStoreHealth(s, users) } + return { key: `${ns}/${name}`, namespace: ns, name, destination: getCNPGObjectStoreDestination(s), users, health: storeHealth(s, users) } }).filter((s) => !clusterFilter || s.users.some((u) => `${u.namespace}/${u.name}` === clusterFilter)) }, [data.objects.objectStores, fleet.rows, clusterFilter])