diff --git a/.design-sync/previews/AlertBanner.tsx b/.design-sync/previews/AlertBanner.tsx index 5cbaf0e46f..9592fbc15e 100644 --- a/.design-sync/previews/AlertBanner.tsx +++ b/.design-sync/previews/AlertBanner.tsx @@ -49,3 +49,17 @@ export function CustomIcon() { ) } + +export function WithAction() { + return ( +
+ Open Operator →} + className="px-3 py-2" + /> +
+ ) +} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 5e5089b568..b141efce73 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -75,6 +75,11 @@ jobs: - name: Run tests run: go test ./... + - name: Evaluate CNPG PromQL in Prometheus + env: + RADAR_TEST_PROMTOOL_IMAGE: prom/prometheus:v3.5.0 + run: go test ./internal/prometheus -run '^TestCNPGDiskGrowthPromQL$' -count=1 + - name: Run PostgreSQL integration tests env: RADAR_TEST_POSTGRES_DSN: postgres://radar:radar@localhost:5432/radar?sslmode=disable diff --git a/CLAUDE.md b/CLAUDE.md index c7a565e8a6..932bf21d03 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -25,6 +25,7 @@ Not everything is in this file. The following files contain critical details tha | Adding or modifying **HTTP endpoints** | `internal/server/server.go` — all routes are defined here — **plus** the handler's doc comments (why the route is gated the way it is lives there; copy the gate of the closest sibling only after reading it) and the integration's section in [docs/integrations.md](docs/integrations.md) | | Adding or modifying **CLI flags** | `cmd/explorer/main.go` — flag definitions and defaults | | Adding a **new CRD integration** (renderer, topology, discovery) | [docs/INTEGRATION_GUIDE.md](docs/INTEGRATION_GUIDE.md) — full checklist with collision gotchas | +| Building a **workspace integration** (several related CRDs with their own screens, like `/cnpg` or `/capacity`) | [docs/INTEGRATION_GUIDE.md](docs/INTEGRATION_GUIDE.md#3-workspace-integrations) — the shared server, UI and action pieces to import, and what is not shared yet; [DESIGN.md](DESIGN.md#unknown-partial-and-denied-values) — how unknown, partial and denied values read | | Working on the **CloudNativePG workspace** (`/cnpg`) | [docs/cnpg.md](docs/cnpg.md) — destinations, navigation (drawer trail, return label, `ctx` guard) and the certainty table: which source each fact comes from and what it reads when unknown. Data from `/api/cnpg/workspace` (per-kind coverage); derivations in `packages/k8s-ui/src/components/cnpg/workspace.ts` + `relations.ts`; screens in `web/src/components/cnpg/` | | Working on **local per-cluster integration settings** (Metrics, Argo CD, Cost in `~/.radar/clusters.json`) | [docs/configuration.md](docs/configuration.md#local-integration-connections) — store `internal/config/profiles.go`, resolve/update `internal/connections`, activation `internal/connectionruntime`, routes `GET/PUT /api/integrations/connections`. In local mode the older `PUT /api/integrations/{prometheus,argocd,cost}` return 409 | | Working on **GitOps** (Argo CD / Flux detail pages, operations, Terminating lifecycle, drift, per-resource health, remote destinations) | [docs/gitops.md](docs/gitops.md) — detail-page tabs, operation semantics, the Terminating severity ramp, nested navigation, single-cluster scope. Engine in `pkg/gitops/`, handlers `internal/server/gitops_handlers.go` | @@ -108,7 +109,7 @@ After `make -demo`, run `kubectl config use-context kind-radar--demo | GitOps | `make gitops-demo` | Argo CD / Flux UI. `-drift` induces live OutOfSync | | Kyverno | `make kyverno-demo` | Policy renderers, report-family selection, admission attribution. Scenarios `openreports` / `modern-only` | | Velero | `make velero-demo` | Backup/restore surfaces — all 13 Backup phases at once. `-live` for states the controller actually produced | -| CloudNativePG | `make cnpg-demo` | CNPG renderers/badges. `-live` for real failovers; fixtures have strict ordering constraints | +| CloudNativePG | `make cnpg-demo` | CNPG renderers/badges. `-live` for real failovers; `-runtime` adds real backups/restore, load, lock chain + Prometheus; fixtures have strict ordering constraints | | Beyla | `make beyla-demo` | `internal/traffic/beyla.go` — which labels exist depends on Beyla config, not code. Modes `attrs` / `no-network` | | Cilium | `make cilium-demo` | `internal/traffic/hubble.go` — every Hubble connection lane. Modes `tls` / `netpol` / `install-radar` | | Kubecost | `make kubecost-demo` | Kubecost 3 current costs — real allocation/assets, local port-forward and in-cluster Service DNS. Modes `query` / `install-radar` / `radar-smoke` | @@ -139,6 +140,7 @@ After `make -demo`, run `kubectl config use-context kind-radar--demo - Helm: `/api/helm/releases/...` - Workloads: `/api/workloads/{kind}/{ns}/{name}/...` (logs, restart, scale, revisions, rollback, images, history). `revisions`/`rollback` accept Deployment, StatefulSet, DaemonSet and **Rollout**, gated by `rollbackableWorkloadKinds` in `server.go`; `images` accepts the same four through a separate map, `workloadImageRoots` in `pkg/k8score/workload_images.go` (a new kind needs both), uses compare-and-swap JSON Patch, and follows a Rollout's `workloadRef`. `history` is the workload's own timeline and never includes sibling workloads — scope rules live in `workload_history.go` and `pkg/timeline`'s `ResourceScope` - Argo Rollouts: `/api/rollouts/{ns}/{name}/{abort,retry,promote,promote-full,skip-step}` (POST) + `/capabilities` (GET); rollback/history deliberately live on the `/workloads` routes. The status verbs patch the `rollouts/status` subresource, so capabilities SAR `rollouts` **and** `rollouts/status` separately — `patch rollouts` does not imply `patch rollouts/status`. Promotion waits for the controller to observe the current pod template, so `promote-full` can return **503** (`ErrControllerNotCaughtUp`) and is safe to retry; for a `workloadRef` Rollout the caller must be able to `get` the referenced workload. Engine `pkg/rollouts`, handlers `internal/server/rollouts_handlers.go` +- GitOps write evidence: `POST /api/gitops/write-evidence` `{kind, group, namespace, name, paths[], owner?}` reads the target (and its GitOps owner) directly as the caller and returns, per field path, whether it appears in last-applied, which field managers own it, and whether an ignore rule covers it, plus the owner's sync policy — never raw managedFields/last-applied. Classification into none/info/may-revert/will-revert is the pure `evaluateGitOpsWriteGuard` in k8s-ui (`utils/gitops-write-guard.ts`); every write dialog shows it via `GitOpsWriteWarning` + `web/src/hooks/useGitOpsWriteGuard.ts`. Copy never promises a change won't be reverted. - GitOps controller actions: `/api/argo/applications/...` (sync, refresh, terminate, suspend, resume, rollback, selective-sync), `/api/flux/{kind}/...` (reconcile, suspend, resume, sync-with-source) - Argo CD API integration: `PUT /api/integrations/argocd` (non-local config only — local mode uses `/api/integrations/connections`; URL/token, probe-before-persist, token preserved across GET-redaction round-trips); `/api/argo/applications/{ns}/{name}/resource-diff` (Git-rendered desired vs live via argocd-server managed-resources; dual RBAC gate + structural Secret redaction; see docs/gitops.md) - GitOps detail data: `/api/gitops/{tree,insights,destination}/{kind}/{ns}/{name}`. `destination` says where a remote Application or `spec.kubeConfig` Flux object deploys; the object and any kubeconfig Secret are read as the caller, and full server URLs never leave the server @@ -153,7 +155,7 @@ After `make -demo`, run `kubectl config use-context kind-radar--demo - RBAC reverse-lookup: `/api/rbac/subject/{kind}/{namespace}/{name}` (ServiceAccount, plus `usedByPods`) and `/api/rbac/subject/{kind}/{name}` (User/Group) — direct + group-inherited bindings and flattened effective rules; `/api/rbac/role/{kind}/{namespace}/{name}` (`_` for a ClusterRole's namespace) — the bindings that reference it; `/api/rbac/namespace/{namespace}` — backs the Namespace RBAC section (group-only ClusterRoleBindings deliberately excluded); `/api/rbac/whoami` — `SelfSubjectRulesReview` pass-through. All gate on `list rolebindings` AND `list clusterrolebindings`: **403 when either is denied, never a silent partial view** - Policy (Kyverno): `/api/policy/resource/{kind}/{ns}/{name}` (one resource's findings), `/api/policy/policies/{policy}` (every resource one policy recorded an outcome for). Report families are authorized **per subject scope** (`policyreports` cluster-wide ≠ `clusterpolicyreports`); findings from an unreadable family are dropped from lists AND counts, with the withheld count reported; `counts` describe the cluster while subject lists are capped and view-filtered. `/api/policy/policies/{policy}/queued` reads Kyverno's `UpdateRequest`s cluster-wide, gated on `list updaterequests` - Velero: `/api/velero/backupstoragelocations/{ns}/{name}/backups` (what a location holds; gated on `list backups`); `POST /api/velero/{backups|restores}/{ns}/{name}/messages` (a run's warnings/errors via a `DownloadRequest` — impersonated; needs a running Velero controller and object storage reachable from Radar, and reports which one failed) -- CloudNativePG: `/api/cnpg/...` — the workspace, operator, catalog reverse-lookups, Cluster logs and activity. Routes, gates and coverage states are listed in [docs/cnpg.md](docs/cnpg.md#api); each is gated on the caller's own access, and a partial answer says what it withheld +- CloudNativePG: `/api/cnpg/...` — the workspace, operator status and diagnosis, catalog reverse-lookups, and per Cluster: runtime, storage, HA facts, history, logs, activity, sessions, restore, report bundle and actions (plus Pooler and ScheduledBackup actions). Routes, gates and coverage states are listed in [docs/cnpg.md](docs/cnpg.md#api); each is gated on the caller's own access, a partial answer says what it withheld, and every write binds the facts the user reviewed ## Key Patterns diff --git a/DESIGN.md b/DESIGN.md index 0241ad5e62..8489e617e1 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -100,6 +100,7 @@ Standard Tailwind type scale. No custom sizes or tracking. Use Tailwind utilitie | `.btn-brand` | Primary CTAs — brand-colored bg, white text, 10px radius | | `.btn-brand-muted` | Secondary brand actions — dimmed brand bg, white text | | `.btn-brand-toggle` | Toggle buttons — 50% brand bg, primary text | +| `.btn-secondary` | Secondary actions beside a `.btn-brand` — bordered surface bg, primary text, 10px radius | Hover/disabled states are built into the classes. For non-brand buttons, use shadcn/ui `} +
Ready Pods from the Service’s EndpointSlices
+ + + + {!spread.known ? ( +
+ + {spread.nodes.length > 0 && ( +
+ Nodes: {spread.nodes.map((n) => `${n.node} (${n.pods.join(', ')})`).join(' · ')} +
+ )} +
+ ) : spread.zones.length === 0 && spread.unlabelled.length === 0 ? ( + + ) : ( +
+
+ {spread.zones.map((z) => ( + + {z.zone} + : {z.pods.join(', ')} + + ))} + {spread.unlabelled.length > 0 && } +
+ {(spread.singleZone || spread.sharedNode) && ( +
+ {spread.singleZone ? 'Every instance is in one zone: losing it loses the cluster. ' : ''} + {spread.sharedNode ? 'Two or more instances share a Node.' : ''} +
+ )} +
topology.kubernetes.io/zone of each instance’s Node
+
+ )} +
+ + {showInstances && + {ha.pods.state !== 'ok' ? ( + + ) : ( +
+ {ha.instances.map((i) => { + const l = liveBy.get(i.pod) + return ( +
+ + + + + + {l?.roleDetail ? CNPG_ROLE_DETAIL_TEXT[l.roleDetail] : i.role === 'unknown' ? 'role unknown' : i.role} + + {l?.timeline !== undefined && timeline {l.timeline}} + {i.qosClass && QoS {i.qosClass}} + {l?.pendingRestart && pending restart} + {i.imageMatches === false && image differs} + {l?.instanceManagerVersion && versions.size > 1 && CNPG instance manager {l.instanceManagerVersion}} +
+ ) + })} + {primaryConflict && } +
+ {live ? 'Role detail and pending restart from each instance manager' : `Role from Pod labels; role detail ${cnpgLiveGap(undefined, liveUnavailable)}`} + {versions.size === 1 ? ` · instance manager ${[...versions][0]}` : ''} +
+
+ )} +
} + + + {!pending.known && pending.pods.length === 0 ? ( + + ) : pending.pods.length === 0 ? ( + None reported + ) : ( + + {pending.pods.join(', ')} {pending.pods.length === 1 ? 'needs' : 'need'} a restart to apply changed parameters + {pending.forDecrease ? ' (a lowered setting: the primary restarts first)' : ''} + {!pending.known ? ` · ${cnpgLiveGap(live, liveUnavailable)}` : ''} + + )} + + + + {!drift.known && drift.drifted.length === 0 ? ( + + ) : drift.drifted.length === 0 ? ( + + {ha.desiredImage} + · observed Pod images match + + ) : ( + + {drift.drifted.map((d) => `${d.pod} has image ${d.image}`).join(' · ')} · desired {ha.desiredImage} + + )} + + + + + + + + + + + + + + + + + + + + {ha.jobs.state !== 'ok' ? ( + + ) : jobs.length === 0 ? ( + None present + ) : ( +
+ {jobs.slice(0, 6).map((j) => ( +
+
+ {j.phase} + {j.role ?? 'job'} + +
+ {j.reason && (/schedul|Too many pods|Insufficient/i.test(j.reason) ?
+ Cannot be scheduled: {summarizeSchedulerMessage(j.reason, { plain: true })} +
{j.reason}
+
: {j.reason})} +
+ ))} + {jobs.length > 6 &&
+{jobs.length - 6} more
} +
+ )} +
+ + + + {showCertificates && } + + ) +} + +/** Certificate expiry and who renews each certificate, folded to one line unless one needs attention. */ +export function CNPGClusterCertificates({ ha, onNavigate }: { ha: CNPGClusterHA; onNavigate?: NavigateToRef }) { + const ns = ha.cluster.namespace + const certs = cnpgCertificateViews(ha.certificates) + const certSummary = cnpgCertificatesSummary(ha.certificates) + return ( + + + + {certs.length === 0 ? ( + + ) : ( +
+ {certs.map((c) => ( +
+ {' '} + + {c.expiresAt ? (c.daysLeft !== undefined && c.daysLeft < 0 ? `expired ${c.expiresAt}` : `expires in ${c.daysLeft} d`) : `expiry unreadable (“${c.raw}”)`} + + + {' · '} + {c.renewal === 'operator' ? 'CloudNativePG renews it' : 'you renew it (spec.certificates)'} + + {c.renewal === 'user' && + (c.certManager ? ( + + {' · cert-manager '} + + + ) : c.metadata?.state === 'ok' ? ( + · not issued by cert-manager + ) : ( + · issuer unknown ({cnpgHASourceText(c.metadata, 'Secret metadata')}) + ))} +
+ ))} +
status.certificates.expirations
+
+ )} +
+
+
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.test.tsx new file mode 100644 index 0000000000..134ea3b0b5 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.test.tsx @@ -0,0 +1,275 @@ +// @vitest-environment jsdom +import { act, type ReactNode } from 'react' +import { createRoot } from 'react-dom/client' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { CNPGClusterSummary, CNPGDimensionMark, CNPGDimensionVerdict, CNPGServingStatus } from './CNPGClusterSummary' +import type { CNPGFleetRow, CNPGProblem } from './workspace' +import type { CNPGDimension } from './ha' +import { OpenIssueContext } from '../problems' + +;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + +function row(over: Partial = {}): CNPGFleetRow { + return { + key: 'db/pg', + namespace: 'db', + name: 'pg', + cluster: { status: { currentPrimary: 'pg-1' } }, + controllerStatus: { text: 'Healthy', level: 'healthy' }, + instances: { ready: 2, desired: 2 }, + pods: [ + { name: 'pg-1', role: 'primary', ready: true }, + { name: 'pg-2', role: 'replica', ready: true }, + ], + replicaCluster: null, + hibernated: false, + pgVersion: '17', + catalog: null, + replication: { text: 'Lag unknown', tone: 'unknown' }, + protection: { + schedule: { text: 'Active', tone: 'healthy', names: [] }, + destination: { text: 'ObjectStore', tone: 'healthy', method: 'plugin' }, + lastSuccessfulBackup: { text: '1 h ago', tone: 'healthy' }, + walArchiving: { text: 'Archiving', tone: 'healthy' }, + recoveryWindow: { text: 'x', tone: 'healthy' }, + restoreValidation: { text: 'None recorded', tone: 'unknown' }, + summary: { text: 'ok', tone: 'healthy' }, + }, + declarations: { summary: { text: 'None', tone: 'neutral' }, total: 0, failed: 0, pending: 0 }, + poolers: [], + poolersKnown: true, + problems: [], + attention: false, + categories: new Set(), + ...over, + } +} + +const problem = (id: string, severity: CNPGProblem['severity'], title: string): CNPGProblem => ({ + id, + severity, + category: 'availability', + title, + subject: { kind: 'Backup', group: 'postgresql.cnpg.io', namespace: 'db', name: `b-${id}` }, + source: 'issue', +}) + +function render(node: ReactNode) { + const host = document.createElement('div') + document.body.appendChild(host) + const root = createRoot(host) + act(() => root.render(node)) + return root +} + +afterEach(() => { + document.body.innerHTML = '' +}) + +describe('CNPGClusterSummary', () => { + it('keeps flat sections by default and frames only At a glance and About when asked', () => { + for (const framed of [false, true]) { + const root = render(Maintenance notice

} operationalFacts={
Base backup progress
} stateFacts={
Existing host facts
} />) + const headings = [...document.querySelectorAll('h3')].filter((h) => ['At a glance', 'About'].includes(h.textContent!)) + expect(headings).toHaveLength(2) + for (const heading of headings) expect(heading.closest('section') !== null).toBe(framed) + const glance = headings[0].parentElement!.parentElement! + const about = headings[1].parentElement!.parentElement! + if (framed) { + expect(glance.textContent).toContain('Base backup progress') + expect(about.textContent).not.toContain('Base backup progress') + expect(about.textContent).toContain('Existing host facts') + expect(glance.textContent).not.toContain('Existing host facts') + expect(glance.textContent).not.toContain('Maintenance notice') + expect(glance.textContent).not.toContain('Backup failed') + } + expect(document.body.textContent!.indexOf('Base backup progress')).toBeLessThan(document.body.textContent!.indexOf('About')) + expect(document.body.textContent!.indexOf('Existing host facts')).toBeGreaterThan(document.body.textContent!.indexOf('About')) + act(() => root.unmount()) + } + const root = render() + expect(document.querySelector('section')).toBeNull() + act(() => root.unmount()) + }) + it('opens the other problems in place when the host links nowhere else', () => { + const onNavigate = vi.fn() + const r = row({ problems: [problem('a', 'critical', 'WAL archiving failing'), problem('b', 'warning', 'Backup failed'), problem('c', 'posture', 'No schedule')], attention: true }) + const root = render() + const more = [...document.querySelectorAll('button')].find((b) => b.textContent?.includes('+2 more'))! + expect(more.getAttribute('aria-expanded')).toBe('false') + act(() => more.click()) + expect(more.getAttribute('aria-expanded')).toBe('true') + expect(document.body.textContent).toContain('Backup failed') + expect(document.body.textContent).toContain('No schedule') + act(() => [...document.querySelectorAll('button')].find((b) => b.textContent === 'b-b')!.click()) + expect(onNavigate).toHaveBeenCalledWith(expect.objectContaining({ kind: 'Backup', name: 'b-b' })) + act(() => root.unmount()) + }) + it('can open with every problem listed', () => { + const r = row({ problems: [problem('a', 'critical', 'WAL archiving failing'), problem('b', 'warning', 'Backup failed')], attention: true }) + const root = render() + const more = [...document.querySelectorAll('button')].find((b) => b.getAttribute('aria-expanded') !== null)! + expect(more.getAttribute('aria-expanded')).toBe('true') + act(() => root.unmount()) + }) + it('keeps a host link as the override', () => { + const r = row({ problems: [problem('a', 'warning', 'One'), problem('b', 'warning', 'Two')], attention: true }) + const root = render( All {n}} />) + expect(document.body.textContent).toContain('All 2') + expect([...document.querySelectorAll('button')].some((b) => b.textContent?.includes('+1 more'))).toBe(false) + act(() => root.unmount()) + }) + it('marks a tab only when its dimension needs a look or could not be assessed', () => { + const dim = (tone: CNPGDimension['tone']): CNPGDimension => ({ id: 'protection', label: 'Backups', tone, text: 'verdict', source: 's' }) + for (const tone of ['healthy', 'neutral'] as const) { + const root = render() + expect(document.querySelector('[aria-label="Backups: verdict"]')).toBeNull() + act(() => root.unmount()) + } + for (const tone of ['degraded', 'unhealthy', 'unknown'] as const) { + const root = render() + expect(document.querySelector('[aria-label="Backups: verdict"]')).not.toBeNull() + act(() => root.unmount()) + } + }) + it("says on the tab why it is marked, and nothing when it is not", () => { + const backups = (tone: CNPGDimension['tone']): CNPGDimension => ({ id: 'protection', label: 'Backups', tone, text: 'no backup destination', source: 'Cluster spec' }) + let root = render() + expect(document.body.textContent).toContain('no backup destination') + expect(document.body.textContent).toContain('Cluster spec') + act(() => root.unmount()) + root = render() + expect(document.body.textContent).toBe('') + act(() => root.unmount()) + }) + it('draws an unassessed dimension as a ring, never a calm dot', () => { + const root = render() + const mark = document.querySelector('[aria-label="Storage: unassessed"]')! + expect(mark.querySelector('.rounded-full.border')).not.toBeNull() + expect(mark.querySelector('.rounded-full:not(.border)')).toBeNull() + act(() => root.unmount()) + }) + it('makes the Serving status a button only when the host can open its details', () => { + const serving: CNPGDimension = { id: 'serving', label: 'Serving', tone: 'healthy', text: 'primary ready', source: 's' } + const onSelect = vi.fn() + let root = render() + expect(document.querySelector('button')).toBeNull() + expect(document.body.textContent).toContain('primary ready') + act(() => root.unmount()) + root = render() + act(() => document.querySelector('[aria-label="Serving: primary ready. Open its details"]')!.click()) + expect(onSelect).toHaveBeenCalled() + act(() => root.unmount()) + }) + it('lists each dimension at a glance, opening its tab', () => { + const dims: CNPGDimension[] = [{ id: 'storage', label: 'Storage', tone: 'degraded', text: 'WAL held by an inactive slot', source: 'slot' }] + const onSelect = vi.fn() + const root = render() + expect(document.body.textContent).toContain('WAL held by an inactive slot') + const open = [...document.querySelectorAll('button')].find((b) => b.textContent === 'Storage →')! + act(() => open.click()) + expect(onSelect).toHaveBeenCalledWith('storage') + act(() => root.unmount()) + }) +}) + +describe('problem meta', () => { + it('keeps a space between the Backup and "and N more"', () => { + const r = row({ + problems: [{ ...problem('a', 'warning', '3 backups failed'), alsoAbout: [{ kind: 'Backup', name: 'b-2' }, { kind: 'Backup', name: 'b-1' }] }], + attention: true, + }) + const root = render( {}} />) + expect(document.body.textContent).toContain('b-a and 2 more') + act(() => root.unmount()) + }) +}) + +describe('problem provenance', () => { + it('names the origin instead of "Radar issue" and links to Issues when the host can', () => { + const open = vi.fn() + const r = row({ + problems: [{ ...problem('a', 'critical', 'WAL archiving is failing'), subject: { kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: 'db', name: 'pg' }, origin: { label: 'Reported by CNPG', detail: 'ContinuousArchiving condition' } }], + attention: true, + }) + const root = render( + + + , + ) + expect(document.body.textContent).toContain('Reported by CNPG') + expect(document.body.textContent).not.toContain('Radar issue') + act(() => [...document.querySelectorAll('button')].find((b) => b.textContent === 'See in Issues →')!.click()) + expect(open).toHaveBeenCalledWith(expect.objectContaining({ id: 'a' })) + act(() => root.unmount()) + }) + it('offers no Issues link without a host handler', () => { + const r = row({ problems: [problem('a', 'warning', 'Backup failed')], attention: true }) + const root = render() + expect(document.body.textContent).toContain('Detected by Radar') + expect(document.body.textContent).not.toContain('See in Issues') + act(() => root.unmount()) + }) +}) + +it('shows the declaration read limitation inline on Overview', () => { + const r = row({ declarations: { total: 1, failed: 0, pending: 0, summary: { text: '≥1 reconciled; Databases not read', tone: 'unknown' } } }) + const root = render() + expect(document.body.textContent).toContain('≥1 reconciled; Databases not read') + act(() => root.unmount()) +}) + +it('shows neutral literal operator phases for every CNPG caller', () => { + const r = row({ cluster: { status: { phase: 'Cluster in healthy state' } } }) + const root = render() + expect(document.body.textContent).toContain('Cluster in healthy state') + const badge = [...document.querySelectorAll('span')].find((el) => el.textContent === 'Cluster in healthy state')! + expect(badge.className).not.toContain('emerald') + act(() => root.unmount()) + const expanded = render() + expect(document.body.textContent).toContain('Cluster in healthy state') + expect(document.body.textContent).not.toContain('Healthy') + act(() => expanded.unmount()) +}) +it('shows healthy Serving only when requested and keeps its tab mark absent', () => { + const dimension: CNPGDimension = { id: 'serving', label: 'Serving', tone: 'healthy', text: 'primary ready', source: 'Pod pg-1 and Service pg-rw' } + const root = render(<>) + expect(document.body.textContent).toContain('primary ready') + expect(document.querySelector('[role="img"]')).toBeNull() + act(() => root.unmount()) + const old = render() + expect(document.body.textContent).not.toContain('primary ready') + act(() => old.unmount()) +}) + +it('keeps literal blocked phases neutral with their explanation and Operator action', () => { + const phase = 'Cluster cannot proceed to reconciliation due to an unknown plugin being required' + const r = row({ cluster: { status: { phase } }, controllerStatus: { text: 'Unknown Plugin', level: 'unhealthy' } }) + const open = vi.fn() + const root = render() + expect(document.body.textContent).toContain(phase) + expect(document.body.textContent).toContain('reported by CNPG') + expect(document.body.textContent).toContain('Plugins this cluster uses') + const badge = [...document.querySelectorAll('span')].find((el) => el.textContent === phase)! + expect(badge.className).not.toMatch(/red|amber|emerald/) + act(() => [...document.querySelectorAll('button')].find((b) => b.textContent === 'Operator and plugins →')!.click()) + expect(open).toHaveBeenCalled() + act(() => root.unmount()) +}) + +it('separates absent operator readiness from observed zero ready Pods', () => { + const root = render() + expect(document.body.textContent).toContain('Not reported by the operator') + expect(document.body.textContent).toContain('0 of 1 instance Pods ready') + expect(document.body.textContent).not.toContain('—/1') + act(() => root.unmount()) +}) + +it('compacts only the scheduler disclosure inside the CNPG problem callout', () => { + const root = render() + const disclosure = [...document.body.querySelectorAll('button')].find((b) => b.textContent?.includes('Scheduler message'))! + const callout = [...document.body.querySelectorAll('div')].find((d) => d.classList.contains('[&_.mt-5]:mt-2')) + expect(callout?.contains(disclosure)).toBe(true) + expect(disclosure.getAttribute('aria-expanded')).toBe('false') + act(() => root.unmount()) +}) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx index a35db6a9bb..7d19d78434 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGClusterSummary.tsx @@ -1,23 +1,32 @@ -import type { ReactNode } from 'react' +import { Fragment, useState, type ReactNode } from 'react' import { clsx } from 'clsx' import { Badge } from '../ui/Badge' import { Tooltip } from '../ui/Tooltip' -import { CNPG_BARMAN_OBJECTSTORE_GROUP, CNPG_GROUP } from '../resources/resource-utils-cnpg' -import type { CNPGFleetRow, CNPGInstance } from './workspace' -import { - FactGrid, - FactRow, - FactSource, - FactValue, - ProblemCallout, - RefLink, - SummaryHeading, - ToneDot, - toneTextClass, - CNPG_PRIMARY_BUTTON, - CNPG_SECONDARY_BUTTON, - type CNPGNavigate, -} from './primitives' +import { Collapse, CollapseChevron, useDisclosure } from '../ui/Collapse' +import { classifyCNPGClusterPhase, cnpgBlockedPhaseExplanation, CNPG_BARMAN_OBJECTSTORE_GROUP, CNPG_GROUP } from '../resources/resource-utils-cnpg' +import { cnpgClusterPlugins, cnpgPluginPhase, cnpgReadyInstances, type CNPGFleetRow, type CNPGInstance } from './workspace' +import type { CNPGDimension } from './ha' +import { PrimaryConflictNote } from './primitives' +import { Note } from './CNPGSharedSummary' +import { type NavigateToRef, RefLink } from '../ui/RefLink' +import { StatusDot, toneTextClass } from '../ui/status-tone' +import { FactGrid, FactRow, FactSource, FactValue, ManagedByText, managedByLabel } from '../facts' +import { ProblemCallout, ProblemList } from '../problems' +import { formatAge } from '../resources/resource-utils' +import { FoldSection, SectionHeading } from '../ui/FoldSection' + +function ReadyCount({ row }: { row: CNPGFleetRow }) { + const r = cnpgReadyInstances(row) + if (row.instances.ready === null) return <>{r.text}{r.podText && {r.podText}} + if (!r.note) return <>{r.text} ready + return ( + + + {r.text} Pods ready status says {row.instances.ready} + + + ) +} export interface CNPGSummaryAction { label: string @@ -25,10 +34,10 @@ export interface CNPGSummaryAction { primary?: boolean } -function InstancePill({ pod, namespace, onNavigate }: { pod: CNPGInstance; namespace: string; onNavigate?: CNPGNavigate }) { +function InstancePill({ pod, namespace, onNavigate }: { pod: CNPGInstance; namespace: string; onNavigate?: NavigateToRef }) { const tone = pod.ready === true ? 'healthy' : pod.ready === false ? 'unhealthy' : 'unknown' const role = pod.role === 'primary' ? 'Primary' : pod.role === 'replica' ? 'Replica' : 'Role unknown' - const readiness = pod.ready === true ? 'Ready' : pod.ready === false ? 'Not ready' : 'Readiness unknown' + const readiness = pod.ready === true ? 'Pod ready' : pod.ready === false ? 'Pod not ready' : 'Pod readiness unknown' return ( @@ -47,122 +56,312 @@ function InstancePill({ pod, namespace, onNavigate }: { pod: CNPGInstance; names ) } +/** + * One dimension's state as a mark beside the tab that explains it: a dot when + * something needs a look, a hollow ring when it could not be assessed (never a + * calm colour), nothing when it is fine. The verdict is in the tooltip; the + * Overview's At a glance has it in words. + */ +export function CNPGDimensionMark({ dimension }: { dimension: CNPGDimension }) { + if (!dimensionMarked(dimension)) return null + const label = `${dimension.label}: ${dimension.text}` + return ( +
{label}
{dimension.source}
} position="bottom"> + + + +
+ ) +} + +function dimensionMarked(dimension: CNPGDimension): boolean { + return dimension.tone !== 'healthy' && dimension.tone !== 'neutral' +} + +function DimensionGlyph({ tone }: { tone: CNPGDimension['tone'] }) { + return tone === 'unknown' ? ( + + ) : ( + + ) +} + +/** + * The verdict behind a tab's mark, as the tab's first line, so a mark always + * points at words on the tab it marks. Nothing when the dimension is fine. + */ +export function CNPGDimensionVerdict({ dimension, className, alwaysShow = false }: { dimension: CNPGDimension; className?: string; alwaysShow?: boolean }) { + if (!alwaysShow && !dimensionMarked(dimension)) return null + return ( +
+ + + + {dimension.label} + {dimension.text} + {dimension.source} +
+ ) +} + +/** Whether the cluster serves writes, on its title line: the headline the tabs do not carry. */ +export function CNPGServingStatus({ dimension, onSelect }: { dimension: CNPGDimension; onSelect?: () => void }) { + const body = ( + <> + + {dimension.label} + {dimension.text} + + ) + const className = 'inline-flex items-center gap-1.5 whitespace-nowrap text-sm' + return ( + + {onSelect ? ( + + ) : ( + {body} + )} + + ) +} + +/** "+N more" that opens the rest of the problems in place, when the host links nowhere else. */ +function MoreProblems({ count, open, onToggle, panelId }: { count: number; open: boolean; onToggle: () => void; panelId: string }) { + return ( + + ) +} + export function CNPGClusterSummary({ row, onNavigate, actions, problemsLink, extra, + lead, + dimensions, + onSelectDimension, + initialProblemsExpanded = false, + stateFacts, + operationalFacts, + dimensionLinkLabel, + onOpenOperator, + framed = false, }: { row: CNPGFleetRow - onNavigate?: CNPGNavigate + onNavigate?: NavigateToRef actions?: CNPGSummaryAction[] /** Link to the complete list of this cluster's findings, shown when more than one exists. */ problemsLink?: (count: number) => ReactNode extra?: ReactNode + /** Rendered first, above the problem callout: standing states such as maintenance mode. */ + lead?: ReactNode + /** Serving · Replication · Storage · Backups, each from its own source (see cnpgDimensions); the At a glance rows. */ + dimensions?: CNPGDimension[] + /** Open with every problem listed below the callout (e.g. arriving from the fleet's "+N more"). */ + initialProblemsExpanded?: boolean + /** Makes each dimension row open where that dimension is explained (its tab). */ + onSelectDimension?: (id: CNPGDimension['id']) => void + /** The name of the place onSelectDimension opens, for the row's link; the dimension's own label when unset. */ + dimensionLinkLabel?: (id: CNPGDimension['id']) => string + /** Opens the operator's own diagnosis, offered beside a controller phase that is not healthy. */ + onOpenOperator?: () => void + /** Extra FactRows appended to About. */ + stateFacts?: ReactNode + /** Extra FactRows appended to At a glance, e.g. live instance operations. */ + operationalFacts?: ReactNode + /** Cards for the full page; drawers keep flat sections. */ + framed?: boolean }) { const top = row.problems[0] const rest = row.problems.length - 1 - const p = row.protection const ns = row.namespace const radarFindings = row.problems.some((x) => x.severity !== 'posture') + const phase = typeof row.cluster?.status?.phase === 'string' ? row.cluster.status.phase : '' + const blocked = classifyCNPGClusterPhase(phase) === 'terminal' ? cnpgBlockedPhaseExplanation(phase, row.cluster?.status?.phaseReason) : null + const [showRest, setShowRest] = useState(initialProblemsExpanded) + const restDisclosure = useDisclosure(showRest) + const Frame = framed ? 'section' : Fragment + const frameProps = framed ? { className: 'mb-4 last:mb-0 rounded-xl border border-theme-border bg-theme-surface px-4 py-3 shadow-theme-sm' } : {} return (
+ {lead} {top && ( - 0 ? problemsLink?.(row.problems.length) ?? +{rest} more : null} - /> + more={ + rest > 0 + ? problemsLink?.(row.problems.length) ?? ( + setShowRest((v) => !v)} panelId={restDisclosure.panelId} /> + ) + : null + } + />
+ )} + {top && rest > 0 && !problemsLink && ( + +
+ +
+
)} {actions && actions.length > 0 && (
{actions.map((a) => ( - ))}
)} - State - - - - - {row.controllerStatus.text} - - - {radarFindings ? 'reported by CNPG · Radar findings above are separate' : 'reported by CNPG'} - - - - -
- - {row.instances.ready ?? '–'}/{row.instances.desired ?? '–'} ready - {row.cluster?.status?.currentPrimary && ( - · primary {row.cluster.status.currentPrimary} + + At a glance + + {dimensions?.map((d) => ( + + onSelectDimension(d.id) : undefined} /> + + ))} + +
+ + + {row.cluster?.status?.currentPrimary && !row.primaryConflict && ( + · primary {row.cluster.status.currentPrimary} + )} + + {row.primaryConflict && } + {row.pods.length > 0 && ( +
+ {row.pods.map((pod) => ( + + ))} +
+ )} +
+
+ + + + {phase || 'Not reported'} + + + {radarFindings ? 'reported by CNPG · Radar findings above are separate' : 'reported by CNPG'} + + {onOpenOperator && (row.controllerStatus.level === 'unhealthy' || row.controllerStatus.level === 'degraded') && ( + )} - {row.pods.length > 0 && ( -
- {row.pods.map((pod) => ( - - ))} + {blocked && ( +
+ {blocked.body} + {cnpgPluginPhase(row.cluster) && + ` Plugins this cluster uses: ${cnpgClusterPlugins(row.cluster).join(', ') || 'none listed'}. The Operator view shows whether each is running and when it last restarted.`}
)} -
-
- - - - - {row.replicaCluster && ( - - Follows {row.replicaCluster.source ? {row.replicaCluster.source} : 'an external primary'} - )} - - {row.pgVersion ?? 'Unknown'} - {row.catalog && ( - - {' · '} - - {row.catalog.name} - - - )} - - - - - - {row.poolers.length === 0 ? ( - - {row.poolersKnown ? 'None' : 'No access to Poolers'} - - ) : ( - - {row.poolers.map((name) => ( - - ))} - + {operationalFacts} +
+ + + + About + + {row.replicaCluster && ( + + Follows {row.replicaCluster.source ? {row.replicaCluster.source} : 'an external primary'} + )} - - {row.gitops && ( - - {row.gitops.tool === 'argocd' ? 'Argo CD' : 'Flux'} {row.gitops.name} + + {row.pgVersion ?? 'Unknown'} + {row.catalog && ( + + {' · '} + + {row.catalog.name} + + + )} + + + + + {row.poolers.length === 0 ? ( + + {row.poolersKnown ? 'None' : 'No access to Poolers'} + + ) : ( + + {row.poolers.map((name) => ( + + ))} + + )} + + {row.managedBy && managedByLabel(row.managedBy) && ( + + + + )} + {stateFacts} + + + + {extra} +
+ ) +} + +function DimensionValue({ dimension: d, linkLabel, onOpen }: { dimension: CNPGDimension; linkLabel: string; onOpen?: () => void }) { + return ( +
+ + + {d.text} + {onOpen && ( + )} - + + {d.source &&
{d.source}
} +
+ ) +} - Protection +/** A cluster's recovery evidence as facts: schedule, destination, newest backup, WAL archiving, recovery window and restore validation. */ +export function CNPGClusterBackupFacts({ row, onNavigate }: { row: CNPGFleetRow; onNavigate?: NavigateToRef }) { + const p = row.protection + const ns = row.namespace + return ( + <> @@ -188,7 +387,7 @@ export function CNPGClusterSummary({ - + {p.recoveryWindow.from ? ( @@ -216,7 +415,23 @@ export function CNPGClusterSummary({ - {extra} - + + ) } + +export function CNPGWALArchivingFact({ fact, compact = false }: { fact: CNPGFleetRow['protection']['walArchiving']; compact?: boolean }) { + const c = fact.operatorCondition + return
+ + + {fact.detail && {compact && fact.state === 'no_destination' ? 'No point-in-time recovery' : fact.detail}} + {c &&
e.stopPropagation()}> + +
{c.type}: {c.status}
+
{c.message || 'No message reported'}
+
Last transition: {c.lastTransitionTime ? : 'Not reported'}
+
+
} +
+} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGConnectSection.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGConnectSection.test.tsx new file mode 100644 index 0000000000..cbe3e234c9 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGConnectSection.test.tsx @@ -0,0 +1,64 @@ +// @vitest-environment jsdom +import { act } from 'react' +import { createRoot } from 'react-dom/client' +import { renderToStaticMarkup } from 'react-dom/server' +import { expect, it, vi } from 'vitest' +import { CNPGConnectSection } from './CNPGConnectSection' + +const cluster = { metadata: { name: 'orders', namespace: 'radar-cnpg-prod' }, spec: { bootstrap: { initdb: { database: 'appdb', owner: 'app' } } } } +;(globalThis as any).IS_REACT_ACT_ENVIRONMENT = true + +it('pairs a namespace-qualified port-forward with matching loopback templates and copies each', async () => { + const writeText = vi.fn().mockResolvedValue(undefined) + Object.defineProperty(navigator, 'clipboard', { value: { writeText }, configurable: true }) + const host = document.createElement('div'); const root = createRoot(host) + act(() => root.render()) + expect(host.textContent).toContain('Inside Kubernetes') + expect(host.textContent).toContain('From this computer') + for (const [label, expected] of [ + ['port-forward command', 'kubectl -n radar-cnpg-prod port-forward service/orders-rw 5432:5432'], + ['local psql command', 'psql -h 127.0.0.1 -p 5432 -U app -d appdb'], + ['local connection string', 'postgresql://app:@127.0.0.1:5432/appdb'], + ]) { + await act(async () => host.querySelector(`[aria-label="Copy ${label}"]`)!.click()) + expect(writeText).toHaveBeenLastCalledWith(expected) + } + act(() => root.unmount()) +}) + +it('uses one three-column grid with every explanation under its own host', () => { + const html = renderToStaticMarkup( {}} />) + const host = document.createElement('div'); host.innerHTML = html + const list = host.querySelector('ul')! + expect(list.className).toContain('grid-cols-[5rem_minmax(0,1fr)_auto]') + expect(list.querySelectorAll('li')).toHaveLength(4) + for (const row of list.querySelectorAll('li')) { + expect(row.className).toBe('contents') + expect(row.children).toHaveLength(3) + expect(row.children[1].className).toContain('[overflow-wrap:anywhere]') + expect(row.children[1].querySelector('div')).not.toBeNull() + expect(row.children[2].textContent).toContain('Reachability') + } +}) + +it('shows command context certainty and per-service availability beside each endpoint', () => { + const host = document.createElement('div') + host.innerHTML = renderToStaticMarkup() + expect(host.textContent).toContain('kubectl --context kind-orders') + expect(host.textContent).not.toContain('uses your current kubectl context') + expect(host.textContent).toContain('Context as named in your kubeconfig (orders-config); kubectl must read the same kubeconfig file.') + expect([...host.querySelectorAll('li')].map((li) => li.textContent)).toEqual([expect.stringContaining('Ready endpoints'), expect.stringContaining('Unavailable: no ready standby'), expect.stringContaining('Ready instance observed')]) + host.innerHTML = renderToStaticMarkup() + expect(host.textContent).toContain('uses your current kubectl context') + expect(host.textContent).not.toContain('--context') + expect([...host.querySelectorAll('li')].every((li) => li.textContent?.includes('Not checked'))).toBe(true) +}) + +it('keeps the unavailable reason beside each unchecked Service', () => { + const host = document.createElement('div') + host.innerHTML = renderToStaticMarkup() + for (const li of host.querySelectorAll('li')) expect(li.textContent).toContain('Not checked · Availability could not be read: timeout') + host.innerHTML = renderToStaticMarkup() + expect(host.querySelector('li')!.textContent).toContain('needs list endpointslices (discovery.k8s.io) in namespace radar-cnpg-prod') + expect(host.querySelectorAll('li')[1].textContent).toContain('needs list pods in namespace radar-cnpg-prod') +}) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGConnectSection.tsx b/packages/k8s-ui/src/components/cnpg/CNPGConnectSection.tsx new file mode 100644 index 0000000000..53a774d274 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGConnectSection.tsx @@ -0,0 +1,179 @@ +import type { CNPGClusterHA } from './ha' +import { useState } from 'react' +import { Check, Copy } from 'lucide-react' +import { Tooltip } from '../ui/Tooltip' +import { CNPG_GROUP } from '../resources/resource-utils-cnpg' +import { CNPG_DEFAULT_PORT, cnpgEndpointAvailability, cnpgConnectionURI, cnpgConnectInfo, cnpgPsqlCommand, cnpgPortForwardCommand, type CNPGConnectEndpoint } from './connect' +import { type NavigateToRef, RefLink } from '../ui/RefLink' +import { FactGrid, FactRow } from '../facts' +import { SectionHeading } from '../ui/FoldSection' + +function CopyButton({ text, label }: { text: string; label: string }) { + const [copied, setCopied] = useState(false) + const copy = () => { + navigator.clipboard?.writeText(text).then( + () => { + setCopied(true) + setTimeout(() => setCopied(false), 2000) + }, + () => {}, + ) + } + return ( + + + + ) +} + +function Snippet({ text, label }: { text: string; label: string }) { + return ( +
+ {text} + +
+ ) +} + +const ROLE_LABEL: Record = { + rw: 'Read-write', + ro: 'Read-only', + r: 'Any instance', + pooler: 'Pooler', + additional: 'Additional', +} + +/** + * How applications reach the cluster: Services, application database and + * owner, and the credentials Secret by name. Never reads the Secret. + */ + +/** + * Selects the Connect heading inside one summary. A data attribute, not an + * id: a drawer summary can sit over a page summary of the same kind, so a host + * scrolls to it within its own summary's element. + */ +export function CNPGConnectSection({ + cluster, + poolers, + poolersKnown, + onNavigate, + onOpenReachability, + showHeading = true, + kubeconfigContext, + kubeconfigSource, + ha, + haUnavailableReason, +}: { + cluster: any + kubeconfigContext?: string + kubeconfigSource?: string + ha?: CNPGClusterHA + haUnavailableReason?: string + poolers?: any[] + /** False when Poolers could not be listed, so a Pooler may exist that is not shown. */ + poolersKnown?: boolean + onNavigate?: NavigateToRef + /** Opens a host's Service on its Reachability tab; the link shows only when given. */ + onOpenReachability?: (service: { namespace: string; name: string }) => void + /** False where the host already titles it (e.g. the Connect dialog). */ + showHeading?: boolean +}) { + const info = cnpgConnectInfo(cluster, poolers) + const ns: string = cluster?.metadata?.namespace ?? '' + const primary = info.endpoints[0] + const local = { ...primary, host: '127.0.0.1', port: CNPG_DEFAULT_PORT } + return ( + <> + {showHeading && Connect} + + +
    + {info.endpoints.map((ep) => { + const availability = cnpgEndpointAvailability(ep, ha, haUnavailableReason) + return ( +
  • + {ROLE_LABEL[ep.role]} +
    + + {`${ep.host}:${ep.port}`} + +
    {ep.selects}
    +
    {availability.text}{availability.source ? ` · ${availability.source}` : ''}
    + {ep.portFromTemplate &&
    port from serviceTemplate
    } +
    + {onOpenReachability ? ( + + + + ) : } +
  • + ) + })} +
+ {info.disabled.length > 0 && ( +
+ Disabled in spec.managed.services: {info.disabled.map((t) => `${cluster?.metadata?.name}-${t}`).join(', ')} +
+ )} + {poolersKnown === false &&
No access to Poolers: one may also front this cluster.
} + {info.replicaCluster && ( +
A replica cluster: every Service reaches instances that only replay until it is promoted.
+ )} +
+ + {info.database.value ? {info.database.value} : Unknown} +
{info.database.source}
+
+ + {info.owner.value ? {info.owner.value} : Unknown} +
{info.owner.source}
+
+ + + + {info.secret.name} + + +
+ {info.secret.source} · Radar does not read it; the password is its password key +
+
+ {info.database.value && ( + +
+ + +
+
+ Via {primary.name}. Replace {''}; psql prompts for it. +
+
+ )} + {info.database.value && ( + +
+ +
{kubeconfigContext ? `Context as named in your kubeconfig${kubeconfigSource ? ` (${kubeconfigSource})` : ''}; kubectl must read the same kubeconfig file.` : 'uses your current kubectl context'}
+ + +
+
Keep port-forward running, then connect in another terminal. Local port 5432 must be free. Replace {''}; psql prompts for it.
+
+ )} +
+ + ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx index dc667382d1..5786604be3 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGDeclarativeSummary.tsx @@ -1,13 +1,15 @@ import type { ReactNode } from 'react' import { getCNPGDeclarativeMessage, getCNPGReclaimPolicy } from '../resources/resource-utils-cnpg' -import type { CNPGWorkspaceResponse } from './workspace' -import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' -import { ClusterLink, NotReported, ObjectProblems, SummaryShell } from './CNPGSharedSummary' +import { cnpgManagedBy, type CNPGWorkspaceResponse } from './workspace' +import { type Fact } from '../facts' +import { cnpgLogicalPaths, type CNPGLogicalPath } from './logicalReplication' +import { CNPGLogicalPathView } from './CNPGLogicalPath' +import { cnpgDatabaseRoleFacts } from './databaseRole' +import { ClusterLink, NotReported, Note, ObjectProblems, SummaryShell } from './CNPGSharedSummary' import { appliedFact, clustersIn, databaseForDeclaration, - gitopsSourceOf, missingManagedRole, observedGenerationFact, refOf, @@ -16,11 +18,15 @@ import { targetCluster, workspaceList, } from './relations' +import { type NavigateToRef, RefLink } from '../ui/RefLink' +import { toneTextClass } from '../ui/status-tone' +import { FactGrid, FactRow, FactValue, ManagedByText, managedByLabel } from '../facts' +import { SectionHeading } from '../ui/FoldSection' interface SummaryProps { resource: any workspace: CNPGWorkspaceResponse | null - onNavigate?: CNPGNavigate + onNavigate?: NavigateToRef } function ReclaimRow({ resource }: { resource: any }) { @@ -41,7 +47,7 @@ function Reconciled({ resource, extra }: { resource: any; extra?: ReactNode }) { const applied = appliedFact(resource) return ( <> - Reconciled + Reconciled @@ -62,14 +68,10 @@ function Reconciled({ resource, extra }: { resource: any; extra?: ReactNode }) { ) } -function DeclaredIn({ resource }: { resource: any }) { - const src = gitopsSourceOf(resource) - if (!src) return - return ( - - {src.tool === 'argocd' ? 'Argo CD application' : 'Flux'} {src.namespace ? `${src.namespace}/${src.name}` : src.name} - - ) +function DeclaredIn({ resource, workspace, onNavigate }: { resource: any; workspace: CNPGWorkspaceResponse | null; onNavigate?: NavigateToRef }) { + const manager = cnpgManagedBy(workspace, resource) + if (!manager || !managedByLabel(manager)) return + return } function DatabaseRef({ resource, workspace, onNavigate }: SummaryProps) { @@ -92,7 +94,7 @@ function DatabaseRef({ resource, workspace, onNavigate }: SummaryProps) { ) } -function LinkList({ items, kind, onNavigate }: { items: any[]; kind: string; onNavigate?: CNPGNavigate }) { +function LinkList({ items, kind, onNavigate }: { items: any[]; kind: string; onNavigate?: NavigateToRef }) { return ( {items.map((o) => ( @@ -114,7 +116,7 @@ export function CNPGDatabaseSummary({ resource, workspace, onNavigate }: Summary - Declared + Declared {resource?.spec?.name ? {resource.spec.name} : } @@ -137,10 +139,10 @@ export function CNPGDatabaseSummary({ resource, workspace, onNavigate }: Summary } /> - Source and target + Source and target - + @@ -149,7 +151,7 @@ export function CNPGDatabaseSummary({ resource, workspace, onNavigate }: Summary {pubsUnavailable ? ( ) : related.publications.length === 0 ? ( - None on this database + No visible Publication declarations ) : ( )} @@ -158,12 +160,13 @@ export function CNPGDatabaseSummary({ resource, workspace, onNavigate }: Summary {subsUnavailable ? ( ) : related.subscriptions.length === 0 ? ( - None on this database + No visible Subscription declarations ) : ( )} + Objects created in SQL are not shown. ) } @@ -191,12 +194,39 @@ function publicationTargets(resource: any): ReactNode { ) } -export function CNPGPublicationSummary({ resource, workspace, onNavigate }: SummaryProps) { +function workspacePaths(workspace: CNPGWorkspaceResponse | null | undefined, subscriptions: any[]): CNPGLogicalPath[] { + return cnpgLogicalPaths(subscriptions, clustersIn(workspace), workspaceList(workspace, 'publications'), workspaceList(workspace, 'poolers'), (ns) => + relationUnavailable(workspace, 'publications', ns, 'Publications'), + ) +} + +export interface CNPGLogicalPathReading { + path: CNPGLogicalPath + /** The publisher primary's report of the slot; absent when not read. */ + slot?: Fact + notice?: ReactNode +} + +export function CNPGPublicationSummary({ + resource, + workspace, + onNavigate, + subscribers, +}: SummaryProps & { + /** Subscriptions reading this publication, with their slots; derived from the workspace when omitted. */ + subscribers?: CNPGLogicalPathReading[] +}) { + const readings: CNPGLogicalPathReading[] = + subscribers ?? + workspacePaths(workspace, workspaceList(workspace, 'subscriptions')) + .filter((p) => p.publication.object?.namespace === resource?.metadata?.namespace && p.publication.object?.name === resource?.metadata?.name) + .map((path) => ({ path })) + const subsUnavailable = relationUnavailable(workspace, 'subscriptions', resource?.metadata?.namespace ?? '', 'Subscriptions') return ( - Declared + Declared {resource?.spec?.name ? {resource.spec.name} : } @@ -210,23 +240,117 @@ export function CNPGPublicationSummary({ resource, workspace, onNavigate }: Summ {publicationTargets(resource)} - + + + Subscribers + {readings.length === 0 ? ( +
+ {subsUnavailable ?? 'No visible Subscription object reads this publication. Subscribers outside Radar\'s view, or created in SQL, are not listed.'} +
+ ) : ( +
+ {readings.map((r) => ( + + ))} +
+ )}
) } -export function CNPGSubscriptionSummary({ resource, workspace, onNavigate }: SummaryProps) { +export function CNPGDatabaseRoleSummary({ resource, workspace, onNavigate }: SummaryProps) { + const ns = resource?.metadata?.namespace ?? '' + const clusterUnavailable = relationUnavailable(workspace, 'clusters', ns, 'Clusters') + const cluster = clusterUnavailable ? null : targetCluster(resource, clustersIn(workspace)) + const f = cnpgDatabaseRoleFacts(resource, cluster) + return ( + + + + Declared + + {f.pgName ? {f.pgName} : } + + + + {f.login ? 'Allowed' : 'Not allowed'}{f.superuser ? ' · superuser' : ''} + + {f.passwordDisabled ? ( + 'Disabled' + ) : f.passwordSecret ? ( + + From Secret {f.passwordSecret} + + ) : ( + No password Secret declared + )} + {f.passwordValidUntil && ( +
+ Valid until {f.passwordValidUntil} (PostgreSQL VALID UNTIL; the operator does not rotate it) +
+ )} +
+ + {f.clientCertificate ? ( + + Operator-issued in Secret {f.clientCertificate.secret} + + {' · '} + {f.clientCertificate.expiration ? `expires ${f.clientCertificate.expiration}` : 'expiry not reported yet'} + + {f.clientCertificate.message &&
{f.clientCertificate.message}
} +
+ ) : ( + Not requested + )} +
+ + + + +
+ + + {f.overriddenByCluster === null ? ( + + ) : f.overriddenByCluster ? ( + + The Cluster declares “{f.pgName}” in spec.managed.roles, which takes precedence: this DatabaseRole is not reconciled while that entry exists + + ) : ( + No spec.managed.roles entry for this role + )} +
+ } + /> + + ) +} + +export function CNPGSubscriptionSummary({ + resource, + workspace, + onNavigate, + logicalPath, +}: SummaryProps & { + /** The path and slot reading; derived from the workspace (slot not read) when omitted. */ + logicalPath?: CNPGLogicalPathReading +}) { + const reading: Partial = logicalPath ?? { path: workspacePaths(workspace, [resource])[0] } const pub = resource?.spec?.publicationName const ext = resource?.spec?.externalClusterName return ( - Declared + Declared {resource?.spec?.name ? {resource.spec.name} : } @@ -254,11 +378,18 @@ export function CNPGSubscriptionSummary({ resource, workspace, onNavigate }: Sum - + + + {reading.path && ( + <> + Replication path + + + )} ) } diff --git a/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx index c7e05904ee..a17740c498 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGImageCatalogSummary.tsx @@ -1,8 +1,11 @@ import { CNPG_GROUP, getCNPGImageCatalogEntries } from '../resources/resource-utils-cnpg' import type { CNPGWorkspaceResponse } from './workspace' -import { FactGrid, FactRow, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' import { NotReported, Note, ObjectProblems, SummaryShell } from './CNPGSharedSummary' import { clustersIn, clustersUsingCatalog, refOf, relationUnavailable } from './relations' +import { type NavigateToRef, RefLink } from '../ui/RefLink' +import { toneTextClass } from '../ui/status-tone' +import { FactGrid, FactRow } from '../facts' +import { SectionHeading } from '../ui/FoldSection' export function CNPGImageCatalogSummary({ resource, @@ -11,7 +14,7 @@ export function CNPGImageCatalogSummary({ }: { resource: any workspace: CNPGWorkspaceResponse | null - onNavigate?: CNPGNavigate + onNavigate?: NavigateToRef }) { const clusterScoped = resource?.kind === 'ClusterImageCatalog' const ns = resource?.metadata?.namespace ?? '' @@ -27,7 +30,7 @@ export function CNPGImageCatalogSummary({ - Images + Images {entries.length === 0 ? (
@@ -42,7 +45,7 @@ export function CNPGImageCatalogSummary({ )} - Used by + Used by {unavailable ? (
diff --git a/packages/k8s-ui/src/components/cnpg/CNPGLogicalPath.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGLogicalPath.test.tsx new file mode 100644 index 0000000000..c5db75ecf9 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGLogicalPath.test.tsx @@ -0,0 +1,22 @@ +import { describe, expect, it } from 'vitest' +import { renderToStaticMarkup } from 'react-dom/server' +import { CNPGLogicalPathView } from './CNPGLogicalPath' +import type { CNPGLogicalPath } from './logicalReplication' + +const path: CNPGLogicalPath = { + subscription: { namespace: 'db', name: 'sub', cluster: 'pg', applied: true }, + externalCluster: { name: 'upstream', declared: true }, + publisher: { kind: 'external', reason: 'the external cluster names no host' }, + publication: { name: 'pub' }, + slot: { name: 'sub' }, + failover: { text: 'unknown', tone: 'unknown' }, +} + +describe('CNPGLogicalPathView', () => { + it('says unknown parts of the path in words, never as "?"', () => { + const html = renderToStaticMarkup() + expect(html).toContain('external cluster upstream') + expect(html).toContain('database unknown') + expect(html).not.toMatch(/upstream<\/span>\/\?|\/\?/) + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGLogicalPath.tsx b/packages/k8s-ui/src/components/cnpg/CNPGLogicalPath.tsx new file mode 100644 index 0000000000..0364ad606a --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGLogicalPath.tsx @@ -0,0 +1,150 @@ +import type { ReactNode } from 'react' +import { ArrowRight } from 'lucide-react' +import { CNPG_GROUP } from '../resources/resource-utils-cnpg' +import { type Fact } from '../facts' +import { cnpgLogicalLocation, type CNPGLogicalPath } from './logicalReplication' +import { type NavigateToRef, RefLink } from '../ui/RefLink' +import { toneTextClass } from '../ui/status-tone' +import { FactGrid, FactRow, FactSource, FactValue } from '../facts' + +// A hop after the first carries its arrow, so a wrapped line never ends on an +// arrow pointing at nothing. +function Hop({ label, children, from }: { label: string; children: ReactNode; from?: boolean }) { + return ( +
+ {from && } +
+
{label}
+
{children}
+
+
+ ) +} + +/** + * One Subscription's path to its publisher: publication, the slot the + * publisher keeps for it, and whether that slot survives a publisher + * failover. `slot` is the host's reading of the publisher primary; absent + * means it was not read. + */ +export function CNPGLogicalPathView({ + path, + slot, + onNavigate, + compact, + notice, +}: { + path: CNPGLogicalPath + slot?: Fact + onNavigate?: NavigateToRef + compact?: boolean + /** The host's word on the slot reading, e.g. that its latest refresh failed. */ + notice?: ReactNode +}) { + const s = path.subscription + const pub = path.publication + const publisher = path.publisher + const slotFact: Fact = slot ?? { text: path.slot.name ? `Slot ${path.slot.name}: not read` : path.slot.reason ?? 'No slot', tone: 'unknown' } + const chain = ( +
+ + + {s.sqlName ?? s.name} + + on {cnpgLogicalLocation(s.cluster ?? 'an unknown cluster', s.dbname)} + + + {pub.object ? ( + + {pub.name} + + ) : pub.name ? ( + {pub.name} + ) : ( + name unknown + )} + + {' '} + on{' '} + {publisher.kind === 'cluster' ? ( + + {publisher.namespace === s.namespace && publisher.name === s.cluster ? 'the same cluster' : `${publisher.namespace}/${publisher.name}`} + + ) : ( + {path.externalCluster.host ?? (path.externalCluster.name ? `external cluster ${path.externalCluster.name}` : 'an unnamed external cluster')} + )} + {pub.dbname ? `/${pub.dbname}` : ' · database unknown'} + + + + + +
+ ) + if (compact) { + return ( +
+ {notice} + {chain} +
Failover: {path.failover.text}
+
+ ) + } + return ( +
+ {notice} + {chain} + + + {publisher.kind === 'cluster' ? ( + + {publisher.namespace}/{publisher.name} via {publisher.via} + + ) : ( + Outside Radar's view: {publisher.reason} + )} +
+ {path.externalCluster.name + ? `From the subscriber's spec.externalClusters[${path.externalCluster.name}].connectionParameters.host` + : "The subscriber's spec.externalClusterName is not set, so no external cluster names the host"} +
+
+ + {pub.object ? ( + + {pub.object.applied === true ? 'Applied' : pub.object.applied === false ? 'Not applied' : 'Pending'} + + ) : ( + + {publisher.kind !== 'cluster' + ? 'Unknown' + : pub.unavailable + ? `Unknown: ${pub.unavailable} in ${publisher.namespace}` + : 'No Publication object declares it: it may exist in SQL only'} + + )} + + + + + + + + + + + {s.applied === false ? ( + Not applied{s.message ? `: ${s.message}` : ''} + ) : s.applied === true ? ( + 'Applied' + ) : ( + Pending + )} +
+ Subscription status. Apply errors and lag are not reported: CloudNativePG's default exporter has no pg_stat_subscription query. +
+
+
+
+ ) +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx index 39e1556ea9..801ea1936d 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectStoreSummary.tsx @@ -8,9 +8,12 @@ import { getCNPGObjectStoreRetention, } from '../resources/resource-utils-cnpg' import type { CNPGWorkspaceResponse } from './workspace' -import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, toneTextClass, type CNPGNavigate } from './primitives' import { NotReported, Note, ObjectProblems, SummaryShell, TimeAgo } from './CNPGSharedSummary' import { clustersIn, inferredObjectStoreHealth, refOf, relationUnavailable, usersOfObjectStore } from './relations' +import { type NavigateToRef, RefLink } from '../ui/RefLink' +import { toneTextClass } from '../ui/status-tone' +import { FactGrid, FactRow, FactValue } from '../facts' +import { SectionHeading } from '../ui/FoldSection' function utc(at: string | undefined): string { if (!at || !Number.isFinite(Date.parse(at))) return 'unknown' @@ -24,7 +27,7 @@ export function CNPGObjectStoreSummary({ }: { resource: any workspace: CNPGWorkspaceResponse | null - onNavigate?: CNPGNavigate + onNavigate?: NavigateToRef }) { const ns = resource?.metadata?.namespace ?? '' const clustersUnavailable = relationUnavailable(workspace, 'clusters', ns, 'Clusters') @@ -46,7 +49,7 @@ export function CNPGObjectStoreSummary({ onNavigate={onNavigate} /> - Upload health + Upload health {clustersUnavailable ? (
@@ -89,7 +92,7 @@ export function CNPGObjectStoreSummary({ )} - Recovery window + Recovery window {windows.length === 0 ? (
@@ -136,7 +139,7 @@ export function CNPGObjectStoreSummary({ )} - Destination + Destination {destination !== '-' ? {destination} : } {provider ?? } @@ -152,7 +155,7 @@ export function CNPGObjectStoreSummary({ {retention ?? } - Used by + Used by {clustersUnavailable ? (
diff --git a/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx index b715f2d584..ede23f0268 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGObjectSummary.test.tsx @@ -2,9 +2,10 @@ import { describe, expect, it } from 'vitest' import { renderToString } from 'react-dom/server' import { CNPGBackupSummary, CNPGScheduledBackupSummary } from './CNPGBackupSummary' import { CNPGObjectStoreSummary } from './CNPGObjectStoreSummary' -import { CNPGDatabaseSummary } from './CNPGDeclarativeSummary' +import { CNPGDatabaseSummary, CNPGPublicationSummary, CNPGSubscriptionSummary } from './CNPGDeclarativeSummary' import { CNPGPoolerSummary } from './CNPGPoolerSummary' import { CNPGImageCatalogSummary } from './CNPGImageCatalogSummary' +import { ClusterLink } from './CNPGSharedSummary' import { CNPG_WORKSPACE_KEYS, type CNPGWorkspaceKey, type CNPGWorkspaceResponse } from './workspace' const PG = 'postgresql.cnpg.io/v1' @@ -77,6 +78,22 @@ describe('CNPGBackupSummary', () => { expect(t).toContain('an earlier schedule of that name') }) + it('distinguishes a replaced Cluster from a missing or unread one without linking to its replacement', () => { + const previous = { ...b, status: { ...b.status, pluginMetadata: { clusterUID: 'old' } } } + const replacement = { ...mainCluster, metadata: { ...mainCluster.metadata, uid: 'new' } } + const html = renderToString() + expect(text(html)).toContain('main · an earlier Cluster of that name; the current one is a different object') + expect(html).not.toContain(')) + expect(missing).toContain('not found in this namespace') + const unread = text(renderToString()) + expect(unread).not.toContain('not found') + expect(unread).not.toContain('earlier Cluster') + const current = renderToString() + expect(current).toContain(' { const issues = [ { id: 'i1', severity: 'critical' as const, kind: 'Backup', group: 'postgresql.cnpg.io', namespace: 'pg', name: 'main-20260901', reason: 'CNPGBackupFailed', message: 'Backup failed' }, @@ -101,6 +118,34 @@ describe('CNPGScheduledBackupSummary', () => { expect(t).toContain('Backups older than 7 days are not listed') }) + it("shows the server's reading and next runs only for the schedule it read", () => { + const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' }, spec: { cluster: { name: 'main' }, schedule: '0 30 2 * * *' } } + const preview = { + schedule: '0 30 2 * * *', + valid: true, + description: 'every day at 02:30:00 UTC', + nextRuns: ['2026-10-01T02:30:00Z', '2026-10-02T02:30:00Z', '2026-10-03T02:30:00Z'], + basis: 'lastCheckTime' as const, + clock: { zone: 'UTC', declared: true, source: 'Operator Deployment declares TZ=UTC' }, + } + const t = text(renderToString()) + expect(t).toContain('every day at 02:30:00 UTC') + expect(t).toContain('2026-10-01 02:30:00 UTC') + expect(t).toContain('Calculated upcoming times') + expect(t).toContain('Next run reported by the operatorNot reported') + expect(t).toContain("operator's last check") + const estimate = text(renderToString()) + expect(estimate).toContain('Estimated upcoming times') + expect(estimate).toContain('UTC estimate') + expect(estimate).not.toContain('Calculated upcoming times') + const stale = text(renderToString()) + expect(stale).not.toContain('every day at') + const due = text(renderToString()) + expect(due).toContain('due (2026-10-01 02:30:00 UTC)') + expect(due).toContain('A run is due on the declared clock') + expect(due).toContain('at most one catch-up backup') + }) + it('says when Backups are not readable instead of listing none', () => { const sched = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' }, spec: { cluster: { name: 'main' } } } const t = text(renderToString()) @@ -196,10 +241,45 @@ describe('CNPGPoolerSummary', () => { it('reports unknown scheduled count and unmeasured pressure', () => { const pooler = { apiVersion: PG, kind: 'Pooler', metadata: { name: 'main-rw', namespace: 'pg' }, spec: { cluster: { name: 'main' }, type: 'rw', instances: 2 } } const t = text(renderToString()) - expect(t).toContain('Scheduled count not reported') + expect(t).toContain('Pooler instance count not reported') expect(t).toContain('Not measured') expect(t).toContain('main-rw') }) + + it('shows Deployment readiness, limits with PgBouncer defaults, observed pause and the Service path when live data is provided', () => { + const pooler = { + apiVersion: PG, + kind: 'Pooler', + metadata: { name: 'main-rw', namespace: 'pg' }, + spec: { cluster: { name: 'main' }, type: 'rw', instances: 2, pgbouncer: { paused: true, parameters: { max_client_conn: '200' } } }, + status: { instances: 2 }, + } + const t = text( + renderToString( + , + ), + ) + expect(t).toContain('from Deployment main-rw') + expect(t).toContain('Pause requested') + expect(t).toContain('1/2 ready') + expect(t).toContain('Pause state') + expect(t).toContain('Observed: Paused on 1 of 2 PgBouncers') + expect(t).toContain('PgBouncer uses 20') + expect(t).toContain('200') + expect(t).toContain('main-rw') + expect(t).toContain('app/app') + expect(t).not.toContain('Not measured') + }) }) describe('CNPGImageCatalogSummary', () => { @@ -234,3 +314,114 @@ describe('CNPGImageCatalogSummary', () => { expect(t).toContain('No access to Clusters') }) }) + +describe('logical replication summaries', () => { + const src = { apiVersion: PG, kind: 'Cluster', metadata: { name: 'src', namespace: 'pg' }, spec: { instances: 2 } } + const dst = { apiVersion: PG, kind: 'Cluster', metadata: { name: 'dst', namespace: 'pg' }, spec: { instances: 1, externalClusters: [{ name: 'src', connectionParameters: { host: 'src-rw', dbname: 'app' } }] } } + const pub = { apiVersion: PG, kind: 'Publication', metadata: { name: 'orders-pub', namespace: 'pg' }, spec: { cluster: { name: 'src' }, name: 'orders_pub', dbname: 'app', target: { allTables: true } }, status: { applied: true } } + const sub = { apiVersion: PG, kind: 'Subscription', metadata: { name: 'orders-sub', namespace: 'pg' }, spec: { cluster: { name: 'dst' }, name: 'orders_sub', dbname: 'app', publicationName: 'orders_pub', externalClusterName: 'src' }, status: { applied: true } } + const w = ws({ clusters: [src, dst], publications: [pub], subscriptions: [sub] }) + + it('shows the subscription path with the slot unread and the failover verdict', () => { + const t = text(renderToString()) + expect(t).toContain('Replication path') + expect(t).toContain('orders_pub') + expect(t).toContain('Slot orders_sub: not read') + expect(t).toContain('Lost on failover') + expect(t).toContain('no pg_stat_subscription query') + }) + + it('never says no Publication declares it when Publications are unreadable', () => { + const denied = ws({ clusters: [src, dst], subscriptions: [sub] }, { coverage: { publications: { state: 'denied' } } }) + const t = text(renderToString()) + expect(t).toContain('Unknown: No access to Publications in pg') + expect(t).not.toContain('No Publication object declares it') + }) + + it('lists the subscribers of a publication', () => { + const t = text(renderToString()) + expect(t).toContain('Subscribers') + expect(t).toContain('orders_sub') + }) +}) + +it('shows the stale declaration as pending in its drawer', () => { + const resource = { apiVersion: PG, kind: 'Database', metadata: { name: 'app', namespace: 'pg', generation: 3 }, spec: { name: 'app' }, status: { applied: true, observedGeneration: 2 } } + const t = text(renderToString()) + expect(t).toContain('Pending · awaiting the operator for the current spec') + expect(t).toContain('AppliedPending') +}) + +it('keeps Deployment readiness beside the pause request even when no Pods are ready', () => { + const resource = { metadata: { name: 'p' }, spec: { pgbouncer: { paused: true } } } + const t = text(renderToString()) + expect(t).toContain('0/2 ready') + expect(t).toContain('Pause requested') + expect(t).not.toContain('Observed: Paused') +}) + +it('puts the pending Pooler cause under readiness and names the Pod whose metric read failed', () => { + const html = renderToString() + const t = text(html) + expect(t).toContain('0/1 ready') + expect(t).toContain('Cannot be scheduled: insufficient cpu.') + expect(t).toContain('Not measured: PgBouncer did not answer') + expect(t).toContain('pooler-pod') + expect(t).toContain('Measurement details') + expect(t).toContain('address not allowed') + expect(t.indexOf('Cannot be scheduled')).toBeGreaterThan(-1) + expect(t.indexOf('Cannot be scheduled')).toBeLessThan(t.indexOf('pooler-pod')) + expect(t.indexOf('Cannot be scheduled')).toBeLessThan(t.indexOf('Connections')) + expect(t.indexOf('address not allowed')).toBeGreaterThan(t.indexOf('Connections')) + expect(html).toContain('button') +}) + +it('never calls an incomplete empty Pooler read idle and names the unread Pod', () => { + const t = text(renderToString()) + expect(t).toContain('No pools seen in what was read') + expect(t).toContain('b: not read (timeout)') + expect(t).not.toContain('Idle:') +}) + +it('leads Pooler observations with the scheduling cause and labels the operator count', () => { + const pod = { pod: 'orders-pooler-pod', state: 'unreachable', reason: 'PgBouncer has not started (Pod cannot be scheduled)', schedulingReason: 'Unschedulable: 0/2 nodes are available: 2 Too many pods. preemption: no victims.' } + const html = renderToString() + const t = text(html) + expect(t).toContain('Pooler status reports 1 instance · 1 requested') + expect(t).toContain('Cannot be scheduled: both nodes have reached their Pod limit') + expect(t).toContain('Not measured: PgBouncer has not started (Pod cannot be scheduled)') + expect(t).toContain('1 not read: orders-pooler-pod (PgBouncer has not started (Pod cannot be scheduled))') + expect(html).toContain('aria-expanded="false"') + expect(t).toContain('preemption: no victims.') +}) + +it('shows the schedule destination blocker only when the target Cluster is readable', () => { + const schedule = { metadata: { name: 'payments-nightly', namespace: 'pg' }, spec: { cluster: { name: 'payments' } } } + const cluster = { apiVersion: PG, kind: 'Cluster', metadata: { name: 'payments', namespace: 'pg' }, spec: {} } + const render = (workspace: CNPGWorkspaceResponse) => text(renderToString()) + const t = render(ws({ clusters: [cluster] })) + expect(t).toContain('Enabled · not run yetNo backup destination') + expect(t).toContain('The resource list status comes from the ScheduledBackup alone.') + expect(render(ws({}))).not.toContain('No backup destination') + expect(render(ws({}))).not.toContain('The resource list status') + expect(render(ws({ clusters: [{ ...cluster, spec: { backup: { barmanObjectStore: { destinationPath: 's3://backups' } } } }] }))).not.toContain('The resource list status') + expect(text(renderToString())).not.toContain('The resource list status') + expect(text(renderToString())).not.toContain('No backup destination') +}) + +it('names visible database declarations without claiming a SQL inventory', () => { + const resource = { metadata: { name: 'orders-appdb', namespace: 'pg' }, spec: { name: 'appdb', cluster: { name: 'orders' } } } + const t = text(renderToString()) + expect(t).toContain('No visible Publication declarations') + expect(t).toContain('No visible Subscription declarations') + expect(t.match(/Objects created in SQL are not shown/g)).toHaveLength(1) + const denied = text(renderToString()) + expect(denied).toContain('No access to Publications') + expect(denied).not.toContain('No visible Publication declarations') +}) + +it.each([['session', 'A client keeps one server connection for its whole session.'], ['transaction', 'A client uses a server connection only for each transaction.']])('explains %s pool mode before its raw value', (poolMode, explanation) => { + const t = text(renderToString()) + expect(t).toContain(explanation) + expect(t.indexOf(explanation)).toBeLessThan(t.indexOf(poolMode, t.indexOf(explanation) + explanation.length)) +}) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGPoolerScheduling.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGPoolerScheduling.test.tsx new file mode 100644 index 0000000000..f20e3955e2 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGPoolerScheduling.test.tsx @@ -0,0 +1,36 @@ +// @vitest-environment jsdom +import { act } from 'react' +import { createRoot } from 'react-dom/client' +import { expect, it, vi } from 'vitest' +import { renderToStaticMarkup } from 'react-dom/server' +import { CNPGPoolerScheduling } from './CNPGPoolerSummary' +Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true }) +it('leads with the cause, truncates a secondary Pod link, exposes its full name on hover and navigates once', () => { + vi.useFakeTimers() + const pod = 'orders-pooler-rw-6f9e349dec-abcdefghij' + const navigate = vi.fn() + const host = document.createElement('div'); document.body.append(host); const root = createRoot(host) + act(() => root.render()) + expect(host.firstElementChild!.firstElementChild!.textContent).toBe('Cannot be scheduled: both nodes have reached their Pod limit.') + const link = [...host.querySelectorAll('button')].find((b) => b.textContent === pod)! + expect(link.parentElement!.className).toContain('[&_button]:truncate') + expect(link.parentElement!.parentElement!.className).toContain('text-theme-text-secondary') + act(() => { link.dispatchEvent(new MouseEvent('mouseover', { bubbles: true })); vi.advanceTimersByTime(350) }) + expect(document.querySelector('[role="tooltip"]')!.textContent).toBe(pod) + act(() => link.click()) + expect(navigate).toHaveBeenCalledExactlyOnceWith({ kind: 'Pod', group: '', namespace: 'db', name: pod }) + act(() => root.unmount()); host.remove(); vi.useRealTimers() +}) + +it('groups a shared scheduling cause while retaining every Pod and its full scheduler message', () => { + const first = '0/2 nodes are available: 2 Too many pods. preemption: no victims.' + const second = '0/2 nodes are available: 2 Too many pods. preemption: no candidates.' + const html = renderToStaticMarkup( {}} />) + expect(html.match(/Cannot be scheduled: both nodes have reached their Pod limit/g)).toHaveLength(1) + expect(html).toContain('Cannot be scheduled: insufficient cpu') + const host = document.createElement('div'); host.innerHTML = html + expect([...host.querySelectorAll('button')].filter((b) => ['a', 'b', 'c'].includes(b.textContent!)).map((b) => b.textContent)).toEqual(['a', 'b', 'c']) + expect(html).toContain(first) + expect(html).toContain(second) + expect(html.match(/Scheduler message/g)).toHaveLength(3) +}) diff --git a/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx index 9418de0626..cebc0e2f52 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGPoolerSummary.tsx @@ -1,8 +1,27 @@ -import { getCNPGPoolerDeploymentName, getCNPGPoolerMode, getCNPGPoolerStatus, isCNPGPoolerPaused } from '../resources/resource-utils-cnpg' +import { Tooltip } from '../ui/Tooltip' +import type { ReactNode } from 'react' +import { getCNPGPoolerDeploymentName, getCNPGPoolerMode, isCNPGPoolerPaused } from '../resources/resource-utils-cnpg' import type { CNPGWorkspaceResponse } from './workspace' -import { FactGrid, FactRow, FactValue, RefLink, SummaryHeading, type CNPGNavigate } from './primitives' import { ClusterLink, NotReported, Note, ObjectProblems, PhaseBadge, SummaryShell } from './CNPGSharedSummary' import { refOf } from './relations' +import { + POOLER_LIMIT_PARAMETERS, + aggregatePoolerPools, + poolerPodPressure, + poolerPressureCoverage, + poolerPressureFact, + observedPause, + poolerBackendService, + poolerReadiness, + type CNPGPoolerLive, + type CNPGPoolerPoolRow, +} from './pooler' +import { type NavigateToRef, RefLink } from '../ui/RefLink' +import { toneTextClass } from '../ui/status-tone' +import { FactGrid, FactRow, FactValue } from '../facts' +import { Badge } from '../ui/Badge' +import { summarizeSchedulerMessage } from '../resources/resource-utils' +import { SectionHeading, FoldSection } from '../ui/FoldSection' const TYPE_LABEL: Record = { rw: 'rw · routes to the primary', @@ -14,36 +33,53 @@ export function CNPGPoolerSummary({ resource, workspace, onNavigate, + live, + actions, + lead, }: { resource: any workspace: CNPGWorkspaceResponse | null - onNavigate?: CNPGNavigate + onNavigate?: NavigateToRef + /** Live reads a host adds (Deployment readiness, PgBouncer metrics and state). */ + live?: CNPGPoolerLive + /** Operations rendered beside the paused state (pause / resume). */ + actions?: ReactNode + /** Rendered first, e.g. a host's notice that a live read is stale. */ + lead?: ReactNode }) { const ns = resource?.metadata?.namespace ?? '' const type = resource?.spec?.type const desired = resource?.spec?.instances - const scheduled = resource?.status?.instances + const reported = resource?.status?.instances const deployment = getCNPGPoolerDeploymentName(resource) + const paused = isCNPGPoolerPaused(resource) + const readiness = poolerReadiness(live?.deployment) + const observed = observedPause(live?.observed) return ( + {lead} - State + State - - {isCNPGPoolerPaused(resource) && PgBouncer is paused: it holds client connections instead of serving them.} +
+ + {paused && Pause requested} +
+ {readiness.detail && {readiness.detail}} +
- {typeof scheduled === 'number' ? `${scheduled} scheduled` : } + {typeof reported === 'number' ? `Pooler status reports ${reported} ${reported === 1 ? 'instance' : 'instances'}` : } {' · '} - {typeof desired === 'number' ? `${desired} desired` : 'desired not set'} + {typeof desired === 'number' ? `${desired} requested` : 'requested count not set'} - The Pooler counts scheduled pods, not ready ones; readiness is on its Deployment. + status.instances is the operator’s count; observed readiness is on its Deployment and Pods. {deployment ? ( @@ -52,19 +88,210 @@ export function CNPGPoolerSummary({ )} - - + +
+ {paused ? 'Pause requested' : 'Serving requested (not paused)'} + {actions} +
+ spec.pgbouncer.paused is what was asked for; each PgBouncer applies it with PAUSE / RESUME. + {paused && When PgBouncer applies the pause, it holds client connections instead of serving them.} + {observed && ( +
+ +
+ )} +
+
+ + Connections + {live?.pressure ? ( + + ) : ( + + + + + + )} + + Limits + + + {getCNPGPoolerMode(resource) === 'transaction' ? 'A client uses a server connection only for each transaction.' : getCNPGPoolerMode(resource) === 'session' ? 'A client keeps one server connection for its whole session.' : 'A client uses a server connection for each statement.'} + {getCNPGPoolerMode(resource)}{!resource?.spec?.pgbouncer?.poolMode ? ' (default)' : ''} + {POOLER_LIMIT_PARAMETERS.map((p) => { + const v = resource?.spec?.pgbouncer?.parameters?.[p.key] + return ( + + {v !== undefined ? ( + {String(v)} + ) : ( + + default{p.pgbouncerDefault ? · PgBouncer uses {p.pgbouncerDefault} : null} + + )} + + {p.key} + + + ) + })} - Routing + Routing {type ? TYPE_LABEL[type] ?? type : } - {getCNPGPoolerMode(resource)} + {live?.service && ( + + + + )}
) } + +function PoolerPath({ resource, live, onNavigate }: { resource: any; live: CNPGPoolerLive; onNavigate?: NavigateToRef }) { + const ns = resource?.metadata?.namespace ?? '' + const svc = live.service! + const backend = poolerBackendService(resource?.spec?.cluster?.name, resource?.spec?.type) + const svcState: Record = { missing: 'does not exist', unreadable: 'no access', foreign: 'not controlled by this Pooler' } + return ( +
+
+ Service + {svc.state === 'ok' ? ( + {svc.port ? ` :${svc.port}` : ''}{svc.type ? ` · ${svc.type}` : ''} + ) : ( + · {svcState[svc.state]} + )} +
+
→ PgBouncer ({resource?.metadata?.name})
+
+ → {backend ? : 'the cluster'} + {' '}of Cluster {resource?.spec?.cluster?.name ?? '—'} +
+
+ ) +} + +// Pods take their client connections independently, so one can queue while +// the sum still looks calm. +function PoolerPodPressure({ pods }: { pods: NonNullable['pods'] }) { + return ( + + + + + + + + + + + + {poolerPodPressure(pods).map((r) => + r.state === 'ok' || r.state === 'partial' ? ( + + + + + + + + ) : ( + + + + + ), + )} + +
PgBouncer PodClients activeWaitingServers activeMax wait
+ {r.pod} + {r.state === 'partial' && partial} + p.pod === r.pod), 'clActive')} /> p.pod === r.pod), 'clWaiting')} /> p.pod === r.pod), 'svActive')} /> p.pod === r.pod), 'maxwaitSeconds')} />
{r.pod} + Not read: {r.error ?? r.state} +
+ ) +} + +function PoolerPressure({ pressure }: { pressure: NonNullable }) { + if (pressure.state === 'loading') return
Reading PgBouncer metrics…
+ if (pressure.state === 'denied' || pressure.state === 'error') { + return + } + const { reporting, limitation, empty } = poolerPressureCoverage(pressure.pods) + const rows = aggregatePoolerPools(reporting) + if (reporting.length === 0) { + return + } + return ( +
+ {rows.length === 0 ? ( +
{empty}
+ ) : ( + + + + + + + + + + + + + {rows.map((r: CNPGPoolerPoolRow) => ( + + + + + + + + + ))} + +
PoolModeClients activeWaitingServers active / idle / usedMax wait
{r.database}/{r.user}{r.poolModes.join(', ') || '—'} / /
+ )} + + Summed over {reporting.length} of {pressure.pods.length} PgBouncer Pods{limitation ? ` · ${limitation}` : ''}. PgBouncer’s admin and + authentication pools are excluded. + + {pressure.pods.length > 1 && } +
+ ) +} + +export function CNPGPoolerScheduling({ namespace, pods, onNavigate }: { namespace: string; pods: NonNullable['pods']; onNavigate?: NavigateToRef }) { + const byCause = new Map() + for (const pod of pods) { + if (!pod.schedulingReason) continue + const cause = summarizeSchedulerMessage(pod.schedulingReason, { plain: true }) + const group = byCause.get(cause) ?? [] + group.push(pod) + byCause.set(cause, group) + } + return <>{[...byCause].map(([cause, blocked]) =>
+
Cannot be scheduled: {cause}.
+ {blocked.map((p) =>
+
+
{p.schedulingReason}
+
)} +
)} +} + +export function CNPGPoolerUnmeasured({ pods }: { pods: NonNullable['pods'] }) { + if (pods.length === 0) return
Not measured: no PgBouncer answered
+ return <>{pods.map((p) =>
+ Not measured: {p.state === 'denied' ? 'needs get pods/proxy' : p.reason ?? 'PgBouncer did not answer'} + {p.pod} + {p.error &&
{p.error}
} +
)} +} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx b/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx index 301ee5ad23..183405aab8 100644 --- a/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx +++ b/packages/k8s-ui/src/components/cnpg/CNPGSharedSummary.tsx @@ -3,8 +3,11 @@ import { Badge } from '../ui/Badge' import type { StatusBadge as StatusBadgeValue } from '../resources/resource-utils' import { CNPG_GROUP } from '../resources/resource-utils-cnpg' import type { CNPGWorkspaceIssue, CNPGWorkspaceResponse } from './workspace' -import { FactValue, ProblemCallout, RefLink, type CNPGNavigate } from './primitives' -import { clustersIn, healthSeverity, problemsForObject, relationUnavailable, targetCluster, type CNPGObjectRef } from './relations' +import { healthToSeverity } from '../../utils/badge-colors' +import { clustersIn, problemsForObject, relationUnavailable, targetCluster, type CNPGObjectRef } from './relations' +import { type NavigateToRef, RefLink } from '../ui/RefLink' +import { FactValue } from '../facts' +import { ProblemCallout } from '../problems' const MAX_PROBLEMS = 3 @@ -20,7 +23,7 @@ export function ObjectProblems({ }: { issues: CNPGWorkspaceIssue[] | undefined subject: CNPGObjectRef - onNavigate?: CNPGNavigate + onNavigate?: NavigateToRef }) { const problems = problemsForObject(issues, subject) if (problems.length === 0) return null @@ -30,6 +33,7 @@ export function ObjectProblems({
{shown.map((p, i) => ( + {status.text} ) @@ -71,16 +75,20 @@ export function ClusterLink({ }: { resource: any workspace: CNPGWorkspaceResponse | null - onNavigate?: CNPGNavigate + onNavigate?: NavigateToRef }) { const name = resource?.spec?.cluster?.name if (!name) return const ns = resource?.metadata?.namespace ?? '' - const visible = !!targetCluster(resource, clustersIn(workspace)) + const clusters = clustersIn(workspace) + const visible = !!targetCluster(resource, clusters) + const replaced = !visible && clusters.some((c) => c.metadata?.namespace === ns && c.metadata?.name === name) return ( - - {workspace && !visible && !relationUnavailable(workspace, 'clusters', ns, 'Clusters') && ( + + {replaced ? ( + · an earlier Cluster of that name; the current one is a different object + ) : workspace && !visible && !relationUnavailable(workspace, 'clusters', ns, 'Clusters') && ( · not found in this namespace )} diff --git a/packages/k8s-ui/src/components/cnpg/CNPGWALArchivingFact.test.tsx b/packages/k8s-ui/src/components/cnpg/CNPGWALArchivingFact.test.tsx new file mode 100644 index 0000000000..aaf3bb65a2 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/CNPGWALArchivingFact.test.tsx @@ -0,0 +1,39 @@ +// @vitest-environment jsdom +import { act } from 'react' +import { createRoot } from 'react-dom/client' +import { expect, it, vi } from 'vitest' +import { CNPGWALArchivingFact } from './CNPGClusterSummary' +Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true }) +it('folds the raw condition and exposes relative age with the exact timestamp on hover', () => { + vi.useFakeTimers(); vi.setSystemTime(new Date('2026-10-01T13:00:00Z')) + const stamp = '2026-10-01T12:00:00Z' + const host = document.createElement('div'); document.body.append(host); const root = createRoot(host) + act(() => root.render()) + expect(host.textContent).toContain('WAL is not archived to recovery storage') + const button = host.querySelector('button')! + expect(button.getAttribute('aria-expanded')).toBe('false') + const panel = document.getElementById(button.getAttribute('aria-controls')!)! + expect(panel.firstElementChild!.hasAttribute('inert')).toBe(true) + act(() => button.click()) + expect(button.getAttribute('aria-expanded')).toBe('true') + expect(host.textContent).toContain('ContinuousArchiving: True') + expect(host.textContent).toContain('Continuous archiving is working') + expect(host.textContent).toContain('1h ago') + expect(host.querySelector('time')?.dateTime).toBe(stamp) + act(() => { host.querySelector('time')!.dispatchEvent(new MouseEvent('mouseover', { bubbles: true })); vi.advanceTimersByTime(350) }) + expect(document.body.textContent).toContain(stamp) + act(() => root.unmount()); host.remove(); vi.useRealTimers() +}) + +it('keeps the compact verdict, consequence and per-row operator disclosure', () => { + const host = document.createElement('div'); const root = createRoot(host) + act(() => root.render()) + expect(host.textContent).toContain('Archive destination absent') + expect(host.textContent).toContain('No point-in-time recovery') + expect(host.textContent).not.toContain('because') + const fold = host.querySelector('button')! + expect(fold.textContent).toContain('Operator report') + act(() => fold.click()) + expect(host.textContent).toContain('ContinuousArchiving: True') + act(() => root.unmount()) +}) diff --git a/packages/k8s-ui/src/components/cnpg/backupRuns.test.ts b/packages/k8s-ui/src/components/cnpg/backupRuns.test.ts new file mode 100644 index 0000000000..75302d6453 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/backupRuns.test.ts @@ -0,0 +1,17 @@ +import { describe, expect, it } from 'vitest' +import { cnpgBackupRunInFlight, cnpgBackupRunsInWindow } from './backupRuns' + +const now = Date.parse('2026-10-05T12:00:00Z') +const backup = (phase: string, ageDays: number) => ({ apiVersion: 'postgresql.cnpg.io/v1', metadata: { name: phase }, status: { phase, startedAt: new Date(now - ageDays * 86400000).toISOString() } }) + +describe('backup run window', () => { + it('retains every in-flight phase regardless of age, including an unknown phase', () => { + const runs = ['pending', 'started', 'running', 'finalizing', 'walArchivingFailing', 'new-phase'].map((phase) => backup(phase, 10)) + expect(cnpgBackupRunsInWindow(runs, now)).toHaveLength(runs.length) + expect(runs.every(cnpgBackupRunInFlight)).toBe(true) + }) + it('applies the cutoff only to settled CNPG runs', () => { + const runs = [backup('completed', 10), backup('failed', 10), backup('completed', 2), backup('failed', 1), { ...backup('running', 1), apiVersion: 'velero.io/v1' }] + expect(cnpgBackupRunsInWindow(runs, now).map((r) => r.status.phase)).toEqual(['failed', 'completed']) + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/backupRuns.ts b/packages/k8s-ui/src/components/cnpg/backupRuns.ts new file mode 100644 index 0000000000..0f3a62be47 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/backupRuns.ts @@ -0,0 +1,19 @@ +import { isApiGroup } from '../resources/resource-utils-cnpg' + +export const CNPG_BACKUP_RUN_WINDOW_MS = 7 * 24 * 60 * 60 * 1000 + +export function cnpgBackupRunInFlight(backup: any): boolean { + return backup?.status?.phase !== 'completed' && backup?.status?.phase !== 'failed' +} + +export function cnpgBackupRunTime(backup: any): string | undefined { + return backup?.status?.stoppedAt || backup?.status?.startedAt || backup?.metadata?.creationTimestamp +} + +export function cnpgBackupRunsInWindow(backups: any[], now = Date.now()): any[] { + return backups.filter((b) => { + if (!isApiGroup(b.apiVersion, 'postgresql.cnpg.io')) return false + const time = Date.parse(cnpgBackupRunTime(b) ?? '') + return cnpgBackupRunInFlight(b) || (Number.isFinite(time) && now - time <= CNPG_BACKUP_RUN_WINDOW_MS) + }).sort((a, b) => Date.parse(cnpgBackupRunTime(b) ?? '') - Date.parse(cnpgBackupRunTime(a) ?? '')) +} diff --git a/packages/k8s-ui/src/components/cnpg/connect.test.ts b/packages/k8s-ui/src/components/cnpg/connect.test.ts new file mode 100644 index 0000000000..d3d7c0feda --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/connect.test.ts @@ -0,0 +1,124 @@ +import { cnpgEndpointAvailability } from './connect' +import type { CNPGClusterHA } from './ha' +import { describe, expect, it } from 'vitest' +import { cnpgConnectInfo, cnpgConnectionURI, cnpgPortForwardCommand, cnpgPsqlCommand } from './connect' + +const cluster = (spec: any = {}) => ({ apiVersion: 'postgresql.cnpg.io/v1', kind: 'Cluster', metadata: { name: 'pg', namespace: 'db' }, spec: { instances: 3, ...spec } }) + +describe('cnpgConnectInfo', () => { + it('forwards the actual Service port to the local template port', () => { + const endpoint = { ...cnpgConnectInfo(cluster()).endpoints[0], port: 6543 } + expect(cnpgPortForwardCommand(endpoint, 'db')).toBe('kubectl -n db port-forward service/pg-rw 5432:6543') + expect(cnpgPortForwardCommand(endpoint, 'db', 15432)).toBe('kubectl -n db port-forward service/pg-rw 15432:6543') + }) + it('lists the default Services and names the Secret by convention when the spec does not', () => { + const info = cnpgConnectInfo(cluster()) + expect(info.endpoints.map((e) => `${e.role} ${e.host}:${e.port}`)).toEqual(['rw pg-rw.db.svc:5432', 'ro pg-ro.db.svc:5432', 'r pg-r.db.svc:5432']) + expect(info.database).toEqual({ value: 'app', source: "CloudNativePG's default" }) + expect(info.owner.value).toBe('app') + expect(info.secret).toEqual({ name: 'pg-app', source: 'name by convention (-app)', byConvention: true }) + expect(info.endpoints.find((e) => e.role === 'r')?.selects).toBe('any instance (may reach the primary; not read-only)') + expect(info.endpoints.find((e) => e.role === 'ro')?.selects).toBe('standbys only (read-only)') + const extra = cnpgConnectInfo(cluster({ managed: { services: { additional: [{ selectorType: 'r', serviceTemplate: { metadata: { name: 'pg-any' } } }] } } })) + expect(extra.endpoints.find((e) => e.name === 'pg-any')?.selects).toContain('not read-only') + }) + + it('reads database, owner and Secret the way CloudNativePG resolves them: recovery, pg_basebackup, then initdb', () => { + const info = cnpgConnectInfo( + cluster({ + bootstrap: { + initdb: { database: 'orders', owner: 'orders_owner', secret: { name: 'init-secret' } }, + recovery: { database: 'restored', secret: { name: 'recovery-secret' } }, + }, + }), + ) + expect(info.database).toEqual({ value: 'restored', source: 'spec.bootstrap.recovery.database' }) + expect(info.owner).toEqual({ value: 'orders_owner', source: 'spec.bootstrap.initdb.owner' }) + expect(info.secret).toEqual({ name: 'recovery-secret', source: 'spec.bootstrap.recovery.secret.name', byConvention: false }) + }) + + it('treats a distributed-topology replica (replica.primary names another cluster) as a replica, like the operator', () => { + expect(cnpgConnectInfo(cluster({ replica: { primary: 'pg-east', source: 'pg-east' } })).replicaCluster).toBe(true) + expect(cnpgConnectInfo(cluster({ replica: { primary: 'pg', source: 'pg-east' } })).replicaCluster).toBe(false) + expect(cnpgConnectInfo(cluster({ replica: { enabled: true, source: 'pg-east' } })).replicaCluster).toBe(true) + expect(cnpgConnectInfo(cluster()).replicaCluster).toBe(false) + }) + + it('owner defaults to the database name', () => { + expect(cnpgConnectInfo(cluster({ bootstrap: { initdb: { database: 'orders' } } })).owner).toEqual({ + value: 'orders', + source: "CloudNativePG's default: the database's name", + }) + }) + + it('does not invent a database for a monolithic import', () => { + const info = cnpgConnectInfo(cluster({ bootstrap: { initdb: { import: { type: 'monolith' } } } })) + expect(info.database.value).toBeUndefined() + expect(info.owner.value).toBeUndefined() + }) + + it('drops disabled default Services and adds managed and Pooler Services with their ports', () => { + const info = cnpgConnectInfo( + cluster({ + managed: { + services: { + disabledDefaultServices: ['ro', 'r'], + additional: [{ selectorType: 'rw', serviceTemplate: { metadata: { name: 'pg-lb' }, spec: { type: 'LoadBalancer', ports: [{ port: 6543 }] } } }], + }, + }, + }), + [ + { metadata: { name: 'pg-pooler-ro', namespace: 'db' }, spec: { cluster: { name: 'pg' }, type: 'ro' } }, + { metadata: { name: 'other', namespace: 'db' }, spec: { cluster: { name: 'pg2' } } }, + { metadata: { name: 'pg-pooler', namespace: 'elsewhere' }, spec: { cluster: { name: 'pg' } } }, + ], + ) + expect(info.disabled).toEqual(['ro', 'r']) + expect(info.endpoints.map((e) => `${e.role} ${e.name}:${e.port}${e.portFromTemplate ? '*' : ''}`)).toEqual(['rw pg-rw:5432', 'additional pg-lb:6543*', 'pooler pg-pooler-ro:5432']) + expect(info.endpoints[2].poolerType).toBe('ro') + expect(info.endpoints[1].selects).toBe('the primary (read-write)') + }) + + it('builds templates with a password placeholder only', () => { + const info = cnpgConnectInfo(cluster({ bootstrap: { initdb: { database: 'orders', owner: 'o w' } } })) + expect(cnpgConnectionURI(info.endpoints[0], info)).toBe('postgresql://o%20w:@pg-rw.db.svc:5432/orders') + expect(cnpgPsqlCommand(info.endpoints[0], info)).toBe("psql -h pg-rw.db.svc -p 5432 -U 'o w' -d orders") + const quoted = cnpgConnectInfo(cluster({ bootstrap: { initdb: { database: "it's" } } })) + expect(cnpgPsqlCommand(quoted.endpoints[0], quoted)).toBe(`psql -h pg-rw.db.svc -p 5432 -U 'it'\\''s' -d 'it'\\''s'`) + }) +}) + +it('pins port-forward to a shell-quoted kubeconfig context when supplied', () => { + const ep = cnpgConnectInfo({ metadata: { name: 'orders', namespace: 'prod' } }).endpoints[0] + expect(cnpgPortForwardCommand(ep, 'prod', 5432, 'kind-cnpg')).toBe('kubectl --context kind-cnpg -n prod port-forward service/orders-rw 5432:5432') + expect(cnpgPortForwardCommand(ep, 'prod', 5432, 'test context')).toContain("--context 'test context'") +}) + +it('reads rw endpoints separately from standby and any-instance Pod readiness', () => { + const eps = cnpgConnectInfo({ metadata: { name: 'orders', namespace: 'prod' } }).endpoints + const ha = { rwEndpoints: { state: 'ok', pods: ['orders-1'] }, pods: { state: 'ok' }, instances: [{ pod: 'orders-1', role: 'primary', ready: true }, { pod: 'orders-2', role: 'replica', ready: false }] } as CNPGClusterHA + expect(eps.map((ep) => cnpgEndpointAvailability(ep, ha).text)).toEqual(['Ready endpoints', 'Unavailable: no ready standby', 'Ready instance observed']) + expect(cnpgEndpointAvailability(eps[1], { ...ha, instances: [] }).text).toBe('Unavailable: no ready standby') + expect(eps.map((ep) => cnpgEndpointAvailability(ep).text)).toEqual(['Not checked', 'Not checked', 'Not checked']) + expect(cnpgEndpointAvailability(eps[0], { ...ha, rwEndpoints: { ...ha.rwEndpoints, state: 'denied' } }).text).toBe('Not checked') + expect(cnpgEndpointAvailability(eps[1], { ...ha, pods: { state: 'denied' } }).text).toBe('Not checked') + expect(cnpgEndpointAvailability(eps[1], { ...ha, instances: [{ ...ha.instances[0], role: 'unknown' }] }).text).toBe('Not checked') +}) + +it('keeps Pooler routing faithful for any-instance and unrecognized selectors', () => { + for (const [type, selects] of [['rw', 'the primary'], ['ro', 'the standbys'], ['r', 'any instance (may reach the primary; not read-only)'], ['future', 'selector future']]) { + const ep = cnpgConnectInfo(cluster(), [{ metadata: { name: 'pooler', namespace: 'db' }, spec: { cluster: { name: 'pg' }, type } }]).endpoints.find((e) => e.role === 'pooler')! + expect(ep.poolerType).toBe(type) + expect(ep.selects).toBe(`PgBouncer in front of ${selects}`) + expect(cnpgEndpointAvailability(ep).source).toBe('This Service’s availability has not been read') + } +}) +it('names the denied grant, failed read, unknown role and unread source alongside Not checked', () => { + const eps = cnpgConnectInfo(cluster()).endpoints + const ha = { rwEndpoints: { state: 'denied', grant: { verb: 'list', group: 'discovery.k8s.io', resource: 'endpointslices', namespace: 'db' } }, pods: { state: 'error', reason: 'API timeout' }, instances: [] } as unknown as CNPGClusterHA + expect(cnpgEndpointAvailability(eps[0], ha)).toEqual({ text: 'Not checked', source: 'No access to Service EndpointSlices (needs list endpointslices (discovery.k8s.io) in namespace db)' }) + expect(cnpgEndpointAvailability(eps[1], ha)).toEqual({ text: 'Not checked', source: 'instance Pods could not be read: API timeout' }) + expect(cnpgEndpointAvailability(eps[0], undefined, 'Reading availability…').source).toBe('Reading availability…') + expect(cnpgEndpointAvailability(eps[0], undefined, 'Availability could not be read: timeout').source).toBe('Availability could not be read: timeout') + expect(cnpgEndpointAvailability(eps[1], { ...ha, pods: { state: 'ok' }, instances: [{ role: 'unknown', ready: true }] } as CNPGClusterHA).source).toBe('Ready instance roles were not reported') +}) diff --git a/packages/k8s-ui/src/components/cnpg/connect.ts b/packages/k8s-ui/src/components/cnpg/connect.ts new file mode 100644 index 0000000000..fe3f4df347 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/connect.ts @@ -0,0 +1,155 @@ +import { cnpgHASourceText, type CNPGClusterHA } from './ha' +import { getCNPGClusterIsReplica } from '../resources/resource-utils-cnpg' + +/** The port CloudNativePG gives PostgreSQL and PgBouncer when a Service template sets none. */ +export const CNPG_DEFAULT_PORT = 5432 + +export type CNPGConnectRole = 'rw' | 'ro' | 'r' | 'pooler' | 'additional' + +export interface CNPGConnectEndpoint { + role: CNPGConnectRole + /** The Service name. */ + name: string + host: string + port: number + /** True when the port comes from a Service template rather than CloudNativePG's default. */ + portFromTemplate: boolean + /** What the Service selects: the primary, standbys only, any instance, or through a Pooler. */ + selects: string + /** Pooler `spec.type`, for pooler endpoints. */ + poolerType?: string +} + +export interface CNPGConnectValue { + value?: string + /** The field the value was read from, or why it is inferred. */ + source: string +} + +export interface CNPGConnectInfo { + endpoints: CNPGConnectEndpoint[] + database: CNPGConnectValue + owner: CNPGConnectValue + secret: { name: string; source: string; byConvention: boolean } + /** A replica cluster's services all reach instances that only replay. */ + replicaCluster: boolean + /** Services named in `spec.managed.services.disabledDefaultServices`. */ + disabled: ('ro' | 'r')[] +} + +const SELECTS: Record<'rw' | 'ro' | 'r', string> = { + rw: 'the primary (read-write)', + ro: 'standbys only (read-only)', + // -r selects every instance Pod (cnpg.io/podRole=instance), the primary included. + r: 'any instance (may reach the primary; not read-only)', +} + +const BOOTSTRAP_ORDER = ['recovery', 'pg_basebackup', 'initdb'] as const + +function templatePort(template: any): number | undefined { + const port = template?.spec?.ports?.[0]?.port + return typeof port === 'number' && port > 0 ? port : undefined +} + +function endpoint(role: CNPGConnectRole, name: string, ns: string, selects: string, template?: any): CNPGConnectEndpoint { + const port = templatePort(template) + return { role, name, host: `${name}.${ns}.svc`, port: port ?? CNPG_DEFAULT_PORT, portFromTemplate: port !== undefined, selects } +} + +// Mirrors CloudNativePG's GetApplicationDatabaseName / Owner / SecretName: +// recovery, then pg_basebackup, then initdb, first non-empty wins. +function fromBootstrap(bootstrap: any, pick: (section: any) => string | undefined, field: string): CNPGConnectValue | undefined { + for (const method of BOOTSTRAP_ORDER) { + const v = pick(bootstrap?.[method]) + if (v) return { value: v, source: `spec.bootstrap.${method}.${field}` } + } + return undefined +} + +/** + * How applications reach a CloudNativePG Cluster, derived from its spec and + * the Poolers that front it. Nothing here reads a Secret: the credentials + * Secret is named from the spec or CloudNativePG's `-app` convention. + */ +export function cnpgConnectInfo(cluster: any, poolers: any[] = []): CNPGConnectInfo { + const name: string = cluster?.metadata?.name ?? '' + const ns: string = cluster?.metadata?.namespace ?? '' + const spec = cluster?.spec ?? {} + const bootstrap = spec.bootstrap + + const disabledRaw: unknown[] = spec.managed?.services?.disabledDefaultServices ?? [] + const disabled = (['ro', 'r'] as const).filter((t) => disabledRaw.includes(t)) + const endpoints: CNPGConnectEndpoint[] = [endpoint('rw', `${name}-rw`, ns, SELECTS.rw)] + for (const t of ['ro', 'r'] as const) if (!disabled.includes(t)) endpoints.push(endpoint(t, `${name}-${t}`, ns, SELECTS[t])) + for (const svc of spec.managed?.services?.additional ?? []) { + const svcName = svc?.serviceTemplate?.metadata?.name + if (!svcName) continue + const sel = svc.selectorType as 'rw' | 'ro' | 'r' + endpoints.push(endpoint('additional', svcName, ns, SELECTS[sel] ?? `selector ${sel ?? 'unknown'}`, svc.serviceTemplate)) + } + for (const p of poolers) { + if (p?.metadata?.namespace !== ns || p?.spec?.cluster?.name !== name || !p?.metadata?.name) continue + const type: string = p.spec?.type || 'rw' + endpoints.push({ + ...endpoint('pooler', p.metadata.name, ns, `PgBouncer in front of ${type === 'rw' ? 'the primary' : type === 'ro' ? 'the standbys' : type === 'r' ? 'any instance (may reach the primary; not read-only)' : `selector ${type}`}`, p.spec?.serviceTemplate), + poolerType: type, + }) + } + + const monolith = bootstrap?.initdb?.import?.type === 'monolith' + const database = + fromBootstrap(bootstrap, (s) => s?.database, 'database') ?? + (monolith + ? { source: 'not set: a monolithic import creates no application database' } + : { value: 'app', source: "CloudNativePG's default" }) + const owner = + fromBootstrap(bootstrap, (s) => s?.owner, 'owner') ?? + (database.value ? { value: database.value, source: "CloudNativePG's default: the database's name" } : { source: 'not set' }) + const secretFromSpec = fromBootstrap(bootstrap, (s) => s?.secret?.name, 'secret.name') + const secret = secretFromSpec?.value + ? { name: secretFromSpec.value, source: secretFromSpec.source, byConvention: false } + : { name: `${name}-app`, source: 'name by convention (-app)', byConvention: true } + + return { + endpoints, + database, + owner, + secret, + replicaCluster: cluster ? getCNPGClusterIsReplica(cluster) : false, + disabled, + } +} + +/** A connection URI with the password left as a placeholder; never a real credential. */ +export function cnpgConnectionURI(ep: CNPGConnectEndpoint, info: CNPGConnectInfo): string { + const user = encodeURIComponent(info.owner.value ?? '') + const db = encodeURIComponent(info.database.value ?? '') + return `postgresql://${user}:@${ep.host}:${ep.port}/${db}` +} + +function shellWord(v: string): string { + return /^[A-Za-z0-9_.@%+=:,/-]+$/.test(v) ? v : `'${v.replace(/'/g, `'\\''`)}'` +} + +/** psql prompts for the password; nothing secret is put on the command line. */ +export function cnpgPsqlCommand(ep: CNPGConnectEndpoint, info: CNPGConnectInfo): string { + return `psql -h ${ep.host} -p ${ep.port} -U ${shellWord(info.owner.value ?? '')} -d ${shellWord(info.database.value ?? '')}` +} + +export function cnpgPortForwardCommand(ep: CNPGConnectEndpoint, namespace: string, localPort = CNPG_DEFAULT_PORT, kubeconfigContext?: string): string { + return `kubectl${kubeconfigContext ? ` --context ${shellWord(kubeconfigContext)}` : ''} -n ${shellWord(namespace)} port-forward ${shellWord(`service/${ep.name}`)} ${localPort}:${ep.port}` +} + +export function cnpgEndpointAvailability(ep: CNPGConnectEndpoint, ha?: CNPGClusterHA, unreadReason?: string): { text: string; source?: string } { + if (ep.role === 'rw' && ha?.rwEndpoints.state === 'ok') { + return { text: ha.rwEndpoints.pods.length > 0 ? 'Ready endpoints' : 'Unavailable: no ready endpoint', source: 'Service EndpointSlices' } + } + if ((ep.role === 'ro' || ep.role === 'r') && ha?.pods.state === 'ok') { + const candidates = ep.role === 'ro' ? ha.instances.filter((i) => i.role === 'replica') : ha.instances + if (ep.role === 'ro' && ha.instances.some((i) => i.ready && i.role === 'unknown') && !candidates.some((i) => i.ready)) return { text: 'Not checked', source: 'Ready instance roles were not reported' } + return { text: candidates.some((i) => i.ready) ? 'Ready instance observed' : ep.role === 'ro' ? 'Unavailable: no ready standby' : 'Unavailable: no ready instance', source: 'Instance Pod readiness' } + } + const source = ep.role === 'rw' ? ha?.rwEndpoints : ep.role === 'ro' || ep.role === 'r' ? ha?.pods : undefined + const what = ep.role === 'rw' ? 'Service EndpointSlices' : 'instance Pods' + return { text: 'Not checked', source: ep.role === 'pooler' || ep.role === 'additional' ? 'This Service’s availability has not been read' : source ? cnpgHASourceText(source, what) : unreadReason ?? 'HA evidence has not been read' } +} diff --git a/packages/k8s-ui/src/components/cnpg/databaseRole.test.ts b/packages/k8s-ui/src/components/cnpg/databaseRole.test.ts new file mode 100644 index 0000000000..5ef6fb0bbd --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/databaseRole.test.ts @@ -0,0 +1,38 @@ +import { describe, expect, it } from 'vitest' +import { cnpgDatabaseRoleFacts, cnpgDatabaseRoleMeta } from './databaseRole' + +const role = (spec: any, status: any = {}) => ({ + apiVersion: 'postgresql.cnpg.io/v1', + kind: 'DatabaseRole', + metadata: { name: 'app-reader', namespace: 'db' }, + spec: { cluster: { name: 'pg' }, name: 'reader', ...spec }, + status, +}) + +describe('cnpgDatabaseRoleFacts', () => { + it('reports the Cluster’s managed.roles precedence as three-valued', () => { + const cluster = { spec: { managed: { roles: [{ name: 'reader' }] } } } + expect(cnpgDatabaseRoleFacts(role({}), cluster).overriddenByCluster).toBe(true) + expect(cnpgDatabaseRoleFacts(role({}), { spec: {} }).overriddenByCluster).toBe(false) + expect(cnpgDatabaseRoleFacts(role({}), null).overriddenByCluster).toBeNull() + }) + + it('reads applied three ways and the operator message', () => { + expect(cnpgDatabaseRoleFacts(role({}), null).state).toBe('pending') + const failed = cnpgDatabaseRoleFacts(role({}, { applied: false, message: 'database role is already managed by the CNPG cluster' }), null) + expect(failed.state).toBe('failed') + expect(failed.message).toContain('already managed') + }) + + it('treats an omitted login as false and names the client certificate Secret', () => { + const f = cnpgDatabaseRoleFacts(role({ clientCertificate: {}, validUntil: '2027-01-01T00:00:00Z' }, { clientCertificate: { expiration: '2026-12-01T00:00:00Z' } }), null) + expect(f.login).toBe(false) + expect(f.clientCertificate?.secret).toBe('app-reader-client-cert') + expect(cnpgDatabaseRoleMeta(f)).toBe('no login · password valid until 2027-01-01T00:00:00Z · client cert until 2026-12-01T00:00:00Z') + }) +}) + +it('keeps a DatabaseRole pending until its current generation is observed', () => { + const r = { ...role({}, { applied: true, observedGeneration: 1 }), metadata: { name: 'reader', generation: 2 } } + expect(cnpgDatabaseRoleFacts(r, null).state).toBe('pending') +}) diff --git a/packages/k8s-ui/src/components/cnpg/databaseRole.ts b/packages/k8s-ui/src/components/cnpg/databaseRole.ts new file mode 100644 index 0000000000..5c79e40e59 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/databaseRole.ts @@ -0,0 +1,80 @@ +import { appliedFact } from './relations' + +// CloudNativePG DatabaseRole (1.30+): one PostgreSQL role declared as its own +// object. The same role name in the Cluster's spec.managed.roles always wins: +// the operator does not reconcile the DatabaseRole and reports it not applied +// ("database role is already managed by the CNPG cluster"). + +export type CNPGRoleState = 'applied' | 'failed' | 'pending' + +export interface CNPGDatabaseRoleFacts { + pgName: string + cluster?: string + state: CNPGRoleState + message?: string + /** Absent in the spec means false: the operator's field is omitempty. */ + login: boolean + superuser: boolean + /** spec.validUntil: when the role's password stops being accepted (PostgreSQL VALID UNTIL). */ + passwordValidUntil?: string + passwordSecret?: string + passwordDisabled: boolean + clientCertificate?: { enabled: boolean; secret: string; expiration?: string; message?: string } + reclaimPolicy: 'delete' | 'retain' + /** + * true when the target Cluster declares the same role in spec.managed.roles, + * false when it does not, null when the Cluster is not visible. + */ + overriddenByCluster: boolean | null +} + +export function cnpgRoleState(obj: any): CNPGRoleState { + const fact = appliedFact(obj) + if (fact.tone === 'healthy') return 'applied' + if (fact.tone === 'unhealthy') return 'failed' + return 'pending' +} + +export function cnpgDatabaseRoleFacts(role: any, cluster: any | null | undefined): CNPGDatabaseRoleFacts { + const spec = role?.spec ?? {} + const status = role?.status ?? {} + const pgName: string = spec.name ?? role?.metadata?.name ?? '' + let overriddenByCluster: boolean | null = null + if (cluster) { + const roles: any[] = Array.isArray(cluster?.spec?.managed?.roles) ? cluster.spec.managed.roles : [] + overriddenByCluster = roles.some((r) => r?.name === pgName) + } + const cc = spec.clientCertificate + const ccEnabled = !!cc && cc.enabled !== false + return { + pgName, + cluster: spec.cluster?.name, + state: cnpgRoleState(role), + message: status.message || undefined, + login: spec.login === true, + superuser: spec.superuser === true, + passwordValidUntil: spec.validUntil || undefined, + passwordSecret: spec.passwordSecret?.name || undefined, + passwordDisabled: spec.disablePassword === true, + clientCertificate: ccEnabled + ? { + enabled: true, + secret: `${role?.metadata?.name}-client-cert`, + expiration: status.clientCertificate?.expiration || undefined, + message: status.clientCertificate?.message || undefined, + } + : undefined, + reclaimPolicy: spec.databaseRoleReclaimPolicy === 'delete' ? 'delete' : 'retain', + overriddenByCluster, + } +} + +/** Short, factual meta line for a list row. */ +export function cnpgDatabaseRoleMeta(f: CNPGDatabaseRoleFacts): string { + const parts: string[] = [] + if (f.overriddenByCluster) parts.push('overridden by spec.managed.roles') + parts.push(f.login ? 'login' : 'no login') + if (f.passwordValidUntil) parts.push(`password valid until ${f.passwordValidUntil}`) + if (f.clientCertificate) parts.push(f.clientCertificate.expiration ? `client cert until ${f.clientCertificate.expiration}` : 'client cert expiry not reported') + return parts.join(' · ') +} diff --git a/packages/k8s-ui/src/components/cnpg/ha.test.ts b/packages/k8s-ui/src/components/cnpg/ha.test.ts new file mode 100644 index 0000000000..47451dc06d --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/ha.test.ts @@ -0,0 +1,426 @@ +import { describe, expect, it } from 'vitest' +import { + cnpgCertificateViews, + cnpgCertificatesSummary, + cnpgHASummary, + cnpgImageDrift, + cnpgDimensions, + cnpgPDBFact, + cnpgLeaseHolderPod, + cnpgLiveGap, + cnpgPendingRestart, + cnpgQuorumFact, + cnpgZoneSpread, + type CNPGClusterHA, + type CNPGHAQuorum, +} from './ha' +import { cnpgLagTone, type CNPGFleetRow } from './workspace' + +function ha(over: Partial = {}): CNPGClusterHA { + return { + cluster: { namespace: 'db', name: 'pg', uid: 'u' }, + sampledAt: '2026-09-30T12:00:00Z', + desiredImage: 'pg:17', + instances: [ + { pod: 'pg-1', podUID: 'a', role: 'primary', ready: true, node: 'n1', zone: 'z1', restartCount: 0, image: 'pg:17', imageMatches: true }, + { pod: 'pg-2', podUID: 'b', role: 'replica', ready: true, node: 'n2', zone: 'z2', restartCount: 0, image: 'pg:17', imageMatches: true }, + ], + pods: { state: 'ok' }, + nodes: { state: 'ok' }, + quorum: { enabled: false, object: { state: 'notFound' } }, + pdbs: { state: 'ok', enabled: true, items: [] }, + primaryLease: { state: 'notFound' }, + operatorLease: { state: 'unavailable' }, + jobs: { state: 'ok', items: [] }, + rwEndpoints: { state: 'ok', service: 'pg-rw', pods: ['pg-1'] }, + certificates: [], + maintenance: { declared: false, inProgress: false, reusePVC: true }, + ...over, + } +} + +function row(over: Partial = {}): CNPGFleetRow { + return { + key: 'db/pg', + namespace: 'db', + name: 'pg', + cluster: { status: { currentPrimary: 'pg-1' } }, + controllerStatus: { text: 'Healthy', level: 'healthy' }, + instances: { ready: 2, desired: 2 }, + pods: [ + { name: 'pg-1', role: 'primary', ready: true }, + { name: 'pg-2', role: 'replica', ready: true }, + ], + replicaCluster: null, + hibernated: false, + pgVersion: '17', + catalog: null, + replication: { text: 'Lag unknown', tone: 'unknown' }, + protection: { + schedule: { text: 'Active', tone: 'healthy', names: [] }, + destination: { text: 'ObjectStore', tone: 'healthy', method: 'plugin' }, + lastSuccessfulBackup: { text: '1 h ago', tone: 'healthy' }, + walArchiving: { text: 'Archiving', tone: 'healthy' }, + recoveryWindow: { text: 'x', tone: 'healthy' }, + restoreValidation: { text: 'None recorded', tone: 'unknown' }, + summary: { text: 'ok', tone: 'healthy' }, + }, + declarations: { summary: { text: 'None', tone: 'neutral' }, total: 0, failed: 0, pending: 0 }, + poolers: [], + poolersKnown: true, + problems: [], + attention: false, + categories: new Set(), + ...over, + } +} + +it('does not call successful snapshot protection WAL archiving when no archive is configured', () => { + const r = row() + r.protection.destination = { text: 'Volume snapshots', tone: 'neutral', method: 'volumeSnapshot' } + r.protection.walArchiving = { text: 'No archive destination configured', tone: 'neutral', source: 'Cluster spec' } + r.protection.lastSuccessfulBackup = { text: 'Completed', tone: 'healthy', source: 'Backup snapshot' } + expect(cnpgDimensions({ row: r }).find((d) => d.id === 'protection')).toMatchObject({ text: 'Backup completed', tone: 'healthy', source: 'Backup snapshot' }) +}) + +describe('cnpgZoneSpread', () => { + it('groups instances by zone and names the primary’s', () => { + const s = cnpgZoneSpread(ha()) + expect(s.known).toBe(true) + expect(s.zones.map((z) => z.zone)).toEqual(['z1', 'z2']) + expect(s.primaryZone).toBe('z1') + expect(s.singleZone).toBe(false) + }) + it('flags every instance in one zone and a shared node', () => { + const s = cnpgZoneSpread( + ha({ + instances: [ + { pod: 'pg-1', podUID: 'a', role: 'primary', ready: true, node: 'n1', zone: 'z1', restartCount: 0 }, + { pod: 'pg-2', podUID: 'b', role: 'replica', ready: true, node: 'n1', zone: 'z1', restartCount: 0 }, + ], + }), + ) + expect(s.singleZone).toBe(true) + expect(s.sharedNode).toBe(true) + }) + it('is unknown, not single-zone, when Nodes are not readable', () => { + const s = cnpgZoneSpread(ha({ nodes: { state: 'denied', grant: { verb: 'get', resource: 'nodes' } } })) + expect(s.known).toBe(false) + expect(s.singleZone).toBe(false) + }) +}) + +describe('cnpgQuorumFact', () => { + const q = (over: Partial): CNPGHAQuorum => ({ enabled: true, object: { state: 'ok' }, ...over }) + it('states R + W > N when it holds', () => { + const f = cnpgQuorumFact(q({ n: 2, w: 1, r: 2, holds: true, status: { standbyNames: ['a', 'b'], standbyNumber: 1 } })) + expect(f.text).toContain('R + W > N') + expect(f.tone).toBe('healthy') + }) + it('states when it does not', () => { + const f = cnpgQuorumFact(q({ n: 2, w: 1, r: 1, holds: false })) + expect(f.text).toContain('R + W ≤ N') + expect(f.tone).toBe('degraded') + }) + it('a reset object is no configuration, not a pass', () => { + expect(cnpgQuorumFact(q({ status: { standbyNames: [], standbyNumber: 0 } })).text).toContain('no synchronous configuration recorded') + }) + it('an unreadable object is unknown', () => { + expect(cnpgQuorumFact(q({ object: { state: 'denied', grant: { verb: 'get', group: 'postgresql.cnpg.io', resource: 'failoverquorums', namespace: 'db' } } })).tone).toBe('unknown') + }) +}) + +describe('cnpgPDBFact', () => { + it('distinguishes disabled from missing and unreadable', () => { + expect(cnpgPDBFact({ state: 'ok', enabled: false, items: [] }).tone).toBe('neutral') + expect(cnpgPDBFact({ state: 'ok', enabled: true, items: [] }).tone).toBe('degraded') + expect(cnpgPDBFact({ state: 'denied', enabled: true, items: [] }).tone).toBe('unknown') + }) +}) + +describe('cnpgCertificateViews', () => { + const now = Date.parse('2026-09-30T00:00:00Z') + const at = (days: number) => new Date(now + days * 86_400_000).toISOString() + it('uses the issues-engine thresholds and owner', () => { + const v = cnpgCertificateViews( + [ + { secret: 'u-20', raw: '', expiresAt: at(20), renewal: 'user' }, + { secret: 'u-3', raw: '', expiresAt: at(3), renewal: 'user' }, + { secret: 'o-3', raw: '', expiresAt: at(3), renewal: 'operator' }, + { secret: 'x', raw: 'garbage', renewal: 'operator' }, + ], + now, + ) + expect(v.map((c) => c.tone)).toEqual(['degraded', 'unhealthy', 'healthy', 'unknown']) + expect(v[0].daysLeft).toBe(20) + }) +}) + +describe('cnpgPendingRestart', () => { + it('is unknown without live data and names the instances otherwise', () => { + expect(cnpgPendingRestart(undefined).known).toBe(false) + const p = cnpgPendingRestart([ + { pod: 'pg-1', state: 'ok', pendingRestart: true, pendingRestartForDecrease: true }, + { pod: 'pg-2', state: 'unreachable' }, + ]) + expect(p.pods).toEqual(['pg-1']) + expect(p.forDecrease).toBe(true) + expect(p.known).toBe(false) + }) + it('does not count an incomplete report as "no restart pending"', () => { + const p = cnpgPendingRestart([ + { pod: 'pg-1', state: 'ok' }, + { pod: 'pg-2', state: 'partial', incomplete: true }, + ]) + expect(p.known).toBe(false) + expect(cnpgPendingRestart([{ pod: 'pg-2', state: 'partial', incomplete: true }]).known).toBe(false) + }) +}) + +describe('cnpgLiveGap', () => { + it('names the missing grant when there is no runtime read', () => { + expect(cnpgLiveGap(undefined, 'needs get pods/proxy in db')).toBe('needs get pods/proxy in db') + }) + it('names each instance that did not report, and why, when runtime access exists', () => { + const gap = cnpgLiveGap([ + { pod: 'pg-1', state: 'ok' }, + { pod: 'pg-2', state: 'unreachable', reason: 'PostgreSQL is not running on this instance' }, + { pod: 'pg-3', state: 'partial', incomplete: true, reason: 'pg_rewind is running' }, + ]) + expect(gap).toBe('pg-2 did not report (PostgreSQL is not running on this instance); pg-3 reported incompletely (pg_rewind is running)') + expect(gap).not.toContain('runtime access') + expect(cnpgLiveGap([{ pod: 'pg-1', state: 'ok' }])).toBe('') + }) +}) + +describe('cnpgDimensions', () => { + it('reads each dimension from its own source and leaves storage unassessed', () => { + const d = cnpgDimensions({ row: row(), ha: ha(), replication: { streaming: 1, standbys: 1, maxReplayLagSeconds: 0 } }) + expect(d.map((x) => [x.id, x.tone])).toEqual([ + ['serving', 'healthy'], + ['replication', 'healthy'], + ['storage', 'unknown'], + ['protection', 'healthy'], + ]) + expect(d.find((x) => x.id === 'protection')?.label).toBe('Backups') + }) + it('storage follows the measured disk fact, and stays unassessed without a measurement', () => { + const measured = cnpgDimensions({ row: row({ disk: { text: '91% used', tone: 'unhealthy', source: 'Fullest: data of pg-1' } }) }) + expect(measured[2]).toMatchObject({ id: 'storage', tone: 'unhealthy', text: '91% used' }) + const unmeasured = cnpgDimensions({ row: row({ disk: { text: 'No usage metrics', tone: 'unknown', source: 'needs Prometheus' } }) }) + expect(unmeasured[2]).toMatchObject({ tone: 'unknown', text: 'No usage metrics: needs Prometheus', source: '' }) + }) + it('names WAL an inactive slot holds even while volume usage is unassessed', () => { + const slot = { id: 'slot:pg/pg:_cnpg_pg_2', reason: 'CNPGInactiveSlot', slot: '_cnpg_pg_2', severity: 'warning', category: 'availability', title: 'Inactive slot _cnpg_pg_2 holds 4.5 GiB of WAL on pg-1 for pg-2', subject: { kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: 'pg', name: 'pg' }, source: 'measurement' } as const + const r = row({ key: 'pg/pg', problems: [slot], disk: { text: 'No usage metrics', tone: 'unknown', source: 'no series' } }) + const d = cnpgDimensions({ row: r }) + expect(d[2]).toMatchObject({ id: 'storage', tone: 'degraded', text: 'WAL held by an inactive slot' }) + expect(d[2].source).toContain('4.5 GiB') + expect(d[2].source).toContain('No usage metrics: no series') + const measured = cnpgDimensions({ row: { ...r, disk: { text: '40% used', tone: 'healthy', source: 'Fullest' } } }) + expect(measured[2]).toMatchObject({ tone: 'degraded', text: '40% used · WAL held by an inactive slot' }) + }) + it('a standby another source saw receiving nothing keeps Replication from reading unassessed or calm', () => { + const gap = { id: 'standby:pg/pg:pg-2', reason: 'CNPGStandbyNotReceiving', severity: 'warning', category: 'availability', title: 'pg-2 is not receiving WAL from the primary', subject: { kind: 'Pod', group: '', namespace: 'pg', name: 'pg-2' }, source: 'measurement' } as const + const r = row({ key: 'pg/pg', problems: [gap] }) + expect(cnpgDimensions({ row: r })[1]).toMatchObject({ tone: 'degraded', text: 'pg-2 not receiving WAL' }) + const live = cnpgDimensions({ row: r, replication: { streaming: 1, standbys: 1, maxReplayLagSeconds: 0 } })[1] + expect(live.tone).toBe('degraded') + expect(live.text).toContain('pg-2 not receiving WAL') + }) + it('counts expected standbys from spec.instances, not from the Pods still running', () => { + // spec.instances 3, only the primary's Pod exists, nothing streams. + const r = row({ instances: { ready: 1, desired: 3 }, pods: [{ name: 'pg-1', role: 'primary', ready: true }] }) + const d = cnpgDimensions({ row: r, replication: { streaming: 0, standbys: 0 } }) + expect(d[1]).toMatchObject({ tone: 'degraded', text: '0 of 2 expected standbys streaming' }) + const one = cnpgDimensions({ row: r, replication: { streaming: 1, standbys: 1, maxReplayLagSeconds: 0 } }) + expect(one[1]).toMatchObject({ tone: 'degraded', text: '1 of 2 expected standbys streaming' }) + }) + it('replication lag takes the same tone as the Replication fact', () => { + const at = (lag: number) => cnpgDimensions({ row: row(), replication: { streaming: 1, standbys: 1, maxReplayLagSeconds: lag } })[1] + expect(at(2)).toMatchObject({ tone: 'healthy' }) + expect(at(8)).toMatchObject({ tone: 'degraded', text: '1 of 1 streaming · replay delay 8.0 s' }) + expect(at(23.6)).toMatchObject({ tone: 'degraded', text: '1 of 1 streaming · replay delay 23 s' }) + expect(at(72)).toMatchObject({ tone: 'unhealthy', text: '1 of 1 streaming · replay delay 72 s' }) + expect(at(72).tone).toBe(cnpgLagTone(72)) + }) + it('a missing standby does not hide a severe lag on the one that streams', () => { + const r = row({ instances: { ready: 2, desired: 3 } }) + const d = cnpgDimensions({ row: r, replication: { streaming: 1, standbys: 1, maxReplayLagSeconds: 72 } })[1] + expect(d).toMatchObject({ tone: 'unhealthy', text: '1 of 2 expected standbys streaming · replay delay 72 s' }) + expect(cnpgDimensions({ row: r, replication: { streaming: 1, standbys: 1, maxReplayLagSeconds: 0.2 } })[1]).toMatchObject({ + tone: 'degraded', + text: '1 of 2 expected standbys streaming', + }) + }) + it('replication is unknown when spec.instances is not reported', () => { + const d = cnpgDimensions({ row: row({ instances: { ready: null, desired: null } }), replication: { streaming: 0, standbys: 0 } }) + expect(d[1].tone).toBe('unknown') + }) + it('replication is unassessed without runtime, never healthy', () => { + const d = cnpgDimensions({ row: row() }) + expect(d.find((x) => x.id === 'replication')?.text).toBe('unassessed') + }) + it('serving fails when the rw endpoint is not on the primary', () => { + const d = cnpgDimensions({ row: row(), ha: ha({ rwEndpoints: { state: 'ok', service: 'pg-rw', pods: ['pg-2'] } }) }) + expect(d[0].tone).toBe('unhealthy') + }) + it('serving is unassessed when instance Pods are not readable', () => { + const d = cnpgDimensions({ row: row({ pods: [] }) }) + expect(d[0].text).toBe('unassessed') + }) +}) + +describe('cnpgLeaseHolderPod', () => { + it('names the Pod of a controller-runtime holder identity', () => { + expect(cnpgLeaseHolderPod('cnpg-controller-manager-5fbdd6bb78-jx82z_6ec0566c-6da6-47aa-9ab1-0e5d3b2c1f11')).toBe('cnpg-controller-manager-5fbdd6bb78-jx82z') + expect(cnpgLeaseHolderPod('pg-1')).toBe('pg-1') + }) +}) + +describe('folded HA and certificates summaries', () => { + const pdbs = { state: 'ok' as const, enabled: true, items: [{ name: 'pg', role: 'replicas', disruptionsAllowed: 1, currentHealthy: 1, expectedPods: 1, observed: true }] } as unknown as CNPGClusterHA['pdbs'] + + it('stays folded with the known facts when nothing is out of line', () => { + const live = [{ pod: 'pg-1', state: 'ok' }, { pod: 'pg-2', state: 'ok' }] as never + expect(cnpgHASummary(ha({ pdbs, operatorLease: { state: 'ok', holder: 'op_1' } }), live)).toEqual({ text: '2/2 observed instances ready · 2 zones · images match', attention: false }) + }) + + it('claims neither readiness nor matching images without instances', () => { + expect(cnpgHASummary(ha({ pdbs, instances: [], operatorLease: { state: 'ok' } }), [] as never)).toEqual({ text: 'no instance Pods', attention: true }) + const unset = ha({ pdbs, operatorLease: { state: 'ok' }, instances: [{ pod: 'pg-1', podUID: 'a', role: 'primary', ready: true, node: 'n1', zone: 'z1', restartCount: 0 }] }) + expect(cnpgHASummary(unset, [{ pod: 'pg-1', state: 'ok' }] as never).text).toBe('1/1 observed instances ready') + }) + + it('names what it could not read instead of reading calm', () => { + const denied = { state: 'denied' as const, grant: { verb: 'list', resource: 'pods', namespace: 'db' } } + const unread = ha({ pods: denied, nodes: denied, pdbs: { ...denied, enabled: true, items: [] }, primaryLease: denied, operatorLease: denied, jobs: { ...denied, items: [] } } as never) + expect(cnpgHASummary(unread, undefined)).toEqual({ + text: 'Not read: Pods, zones, disruption budgets, primary lease, operator lease, Jobs, pending restarts', + attention: false, + }) + expect(cnpgHASummary(ha({ pdbs }), undefined).text).toBe('2/2 observed instances ready · 2 zones · images match · not read: operator lease, pending restarts') + }) + + it('opens and leads with what is wrong', () => { + const shared = ha({ + pdbs, + instances: [ + { pod: 'pg-1', podUID: 'a', role: 'primary', ready: true, node: 'n1', zone: 'z1', restartCount: 0, image: 'pg:17', imageMatches: true }, + { pod: 'pg-2', podUID: 'b', role: 'replica', ready: false, node: 'n1', zone: 'z1', restartCount: 0, image: 'pg:16', imageMatches: false }, + ], + }) + const s = cnpgHASummary(shared, [{ pod: 'pg-2', state: 'ok', pendingRestart: true } as never]) + expect(s.attention).toBe(true) + expect(s.text).toBe('1 of 2 instances not ready · every instance in one zone · instances share a Node · restart pending on pg-2 · an instance Pod has a different image · not read: operator lease') + }) + + it('names the nearest certificate expiry and who renews them', () => { + const now = Date.parse('2026-09-30T00:00:00Z') + const certs = [ + { secret: 'pg-ca', expiresAt: '2026-12-29T00:00:00Z', renewal: 'operator' }, + { secret: 'pg-server', expiresAt: '2026-10-20T00:00:00Z', renewal: 'operator' }, + ] as never + expect(cnpgCertificatesSummary(certs, now)).toEqual({ text: '2 certificates · nearest reported expiry in 20 d (pg-server) · CloudNativePG renews them', attention: false }) + const oneUnread = [{ secret: 'pg-ca', expiresAt: '2026-12-29T00:00:00Z', renewal: 'operator' }, { secret: 'pg-server', raw: 'garbage', renewal: 'operator' }] as never + expect(cnpgCertificatesSummary(oneUnread, now)).toEqual({ text: '2 certificates · nearest reported expiry in 90 d (pg-ca) · 1 expiry unreadable · CloudNativePG renews them', attention: true }) + const userSoon = [{ secret: 'app-tls', expiresAt: '2026-10-05T00:00:00Z', renewal: 'user' }] as never + expect(cnpgCertificatesSummary(userSoon, now).attention).toBe(true) + expect(cnpgCertificatesSummary([], now)).toEqual({ text: 'No expiry reported by the operator', attention: false }) + }) +}) + +describe('replication chip and the sustained-lag finding', () => { + it('reads no calmer than a standby measured far behind for the whole window', () => { + const problem = { id: 'lag:db/pg', reason: 'CNPGSustainedLag', severity: 'critical', category: 'availability', title: 'pg-2 ≥ 24 h behind in every sample for 10 min', subject: { kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: 'db', name: 'pg' }, source: 'measurement' } as never + const dims = cnpgDimensions({ row: row({ problems: [problem] }), replication: { streaming: 1, standbys: 1, maxReplayLagSeconds: 0 } }) + const rep = dims.find((d) => d.id === 'replication')! + expect(rep.tone).toBe('unhealthy') + expect(rep.text).toBe('1 of 1 standbys streaming · sustained lag') + expect(rep.source).toBe('pg-2 ≥ 24 h behind in every sample for 10 min') + const unread = cnpgDimensions({ row: row({ problems: [problem] }) }).find((d) => d.id === 'replication')! + expect(unread).toMatchObject({ tone: 'unhealthy', text: 'sustained lag' }) + }) +}) + +describe('backup dimension certainty', () => { + it('never treats archiving alone as restorable', () => { + const r = row() + r.protection.lastSuccessfulBackup = { text: 'No successful backup yet', tone: 'degraded', source: 'Backups read in this namespace; none completed' } + expect(cnpgDimensions({ row: r }).find((d) => d.id === 'protection')).toMatchObject({ text: 'CNPG reports archiving · no successful backup yet', tone: 'degraded', source: 'Backups read in this namespace; none completed' }) + r.protection.lastSuccessfulBackup = { text: 'No access to Backups', tone: 'unknown', source: 'Backups not read' } + expect(cnpgDimensions({ row: r }).find((d) => d.id === 'protection')).toMatchObject({ text: 'unassessed', tone: 'unknown', source: 'Backups not read' }) + }) +}) + +it('uses declared instances for readiness and opens for an instance Pod not observed', () => { + const base = ha({ declaredInstances: 2, expectedInstances: ['pg-1', 'pg-2'] }) + base.instances = base.instances.slice(0, 1) + const summary = cnpgHASummary(base, [{ pod: 'pg-1', state: 'ok' }]) + expect(summary.attention).toBe(true) + expect(summary.text).toContain('1 of 2 declared instances ready; no instance Pod observed for pg-2') + expect(summary.text).not.toContain('1/1') +}) + +it('names excess observed instances without putting them over a smaller denominator', () => { + const summary = cnpgHASummary(ha({ declaredInstances: 1 }), undefined) + expect(summary.text).toContain('2 observed instances ready; 1 instances declared') + expect(summary.text).not.toContain('2 of 1') +}) +it('does not claim matching images without observed image evidence', () => { + expect(cnpgImageDrift(ha({ instances: [] })).known).toBe(false) + const base = ha() + base.instances[0].imageMatches = undefined + expect(cnpgImageDrift(base).known).toBe(false) +}) + +it('uses a definitive non-serving verdict from an empty complete endpoint read without a primary', () => { + const r = row({ cluster: { status: {} }, pods: [] }) + const empty = ha({ rwEndpoints: { state: 'ok', service: 'analytics-rw', pods: [] } }) + expect(cnpgDimensions({ row: r, ha: empty })[0]).toMatchObject({ tone: 'unhealthy', text: 'not serving: no ready read-write endpoint', source: 'EndpointSlices of Service analytics-rw' }) + expect(cnpgDimensions({ row: r, ha: empty })[0].tone).toBe('unhealthy') + expect(cnpgDimensions({ row: r, ha: ha({ rwEndpoints: { state: 'denied', service: 'analytics-rw', pods: [] } }) })[0].tone).toBe('unknown') + expect(cnpgDimensions({ row: { ...r, hibernated: true }, ha: empty })[0].text).toBe('hibernated') +}) +it('uses storage reasons for loading, missing Prometheus and denied usage', () => { + expect(cnpgDimensions({ row: row() })[2].text).toBe('Reading…') + expect(cnpgDimensions({ row: row({ disk: { tone: 'unknown', text: 'No usage metrics', source: 'Prometheus not connected' } }) })[2].text).toBe('No usage metrics') + expect(cnpgDimensions({ row: row({ disk: { tone: 'unknown', text: 'No access', source: 'Needs list persistentvolumeclaims in namespace db' } }) })[2].text).toContain('Needs list persistentvolumeclaims') +}) + +it('retains the missing standby and failover consequence when streaming is denied', () => { + const h = ha({ expectedInstances: ['pg-1', 'pg-2'], instances: [ha().instances[0]] }) + const d = cnpgDimensions({ row: row(), ha: h, replicationGap: 'needs get pods/proxy in namespace db' }).find((d) => d.id === 'replication')! + expect(d.tone).toBe('degraded') + expect(d.text).toBe('Expected standby pg-2 is not running') + expect(d.source).toContain('streaming not measured: needs get pods/proxy in namespace db') + expect(d.source).toContain('No ready standby to fail over to') + const unread = cnpgDimensions({ row: row(), ha: { ...h, pods: { state: 'denied' } } }).find((d) => d.id === 'replication')! + expect(unread.text).toBe('unassessed') + expect(unread.source).not.toContain('No ready standby') +}) + +it('keeps the Storage verdict concise while the Storage notice owns discovery details', () => { + const r = row({ disk: { tone: 'unknown', text: 'No usage metrics', source: 'Prometheus not connected', detail: 'No working endpoint. Candidate monitoring/prometheus.' } }) + expect(cnpgDimensions({ row: r }).find((d) => d.id === 'storage')).toMatchObject({ text: 'No usage metrics', source: '' }) +}) +it('adds the failover consequence to measured replication without duplicating streaming facts', () => { + const h = ha({ instances: [ha().instances[0]], expectedInstances: ['pg-1', 'pg-2'] }) + const d = cnpgDimensions({ row: row(), ha: h, replication: { streaming: 0, standbys: 0 } }).find((d) => d.id === 'replication')! + expect(d.text).toBe('0 of 1 expected standbys streaming') + expect(d.source).toContain('No ready standby to fail over to') + expect(d.tone).toBe('degraded') +}) + +it('preserves separately measured standby gaps beside missing Pods and avoids inventing a grant', () => { + const h = ha({ expectedInstances: ['pg-1', 'pg-2', 'pg-3'], instances: [ha().instances[0], { ...ha().instances[1], pod: 'pg-3' }] }) + const r = row({ instances: { desired: 3, ready: 2 }, problems: [{ id: 'standby:db/pg:pg-3', reason: 'CNPGStandbyNotReceiving', severity: 'warning', category: 'replication', title: 'pg-3 receiver is down', subject: { kind: 'Pod', name: 'pg-3' }, source: 'measurement' } as any] }) + const d = cnpgDimensions({ row: r, ha: h }).find((d) => d.id === 'replication')! + expect(d.text).toContain('Expected standby pg-2 is not running') + expect(d.text).toContain('pg-3 not receiving WAL') + expect(d.source).toContain('pg-3 receiver is down') + expect(d.source).toContain('streaming not measured: not read') + expect(d.source).not.toContain('needs get pods/proxy') + h.instances = [h.instances[0]] + expect(cnpgDimensions({ row: r, ha: h }).find((d) => d.id === 'replication')!.text).toContain('Expected standbys pg-2, pg-3 are not running') +}) diff --git a/packages/k8s-ui/src/components/cnpg/ha.ts b/packages/k8s-ui/src/components/cnpg/ha.ts new file mode 100644 index 0000000000..bf117736be --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/ha.ts @@ -0,0 +1,602 @@ +// High-availability facts for one CloudNativePG Cluster, from +// GET /api/cnpg/clusters/{ns}/{name}/ha, plus pure derivations the Overview and +// the switchover dialog share. Each sub-read carries its own state: a denied or +// unavailable source is "unknown", never none or healthy. + +import { formatAge, type HealthLevel } from '../resources/resource-utils' +import { formatGrant, type Grant } from '../../utils/grant' +import { CNPG_PROMETHEUS_NOT_CONNECTED, cnpgFormatLag, cnpgLagTone, cnpgReplicationTone, type CNPGFleetRow } from './workspace' +import { type Fact } from '../facts' +import { type FoldSummary } from '../ui/FoldSection' +import { worseTone } from '../ui/status-tone' + +export type CNPGHASourceState = 'ok' | 'denied' | 'notFound' | 'notInstalled' | 'unavailable' | 'error' + +export interface CNPGHASource { + state: CNPGHASourceState + reason?: string + grant?: Grant +} + +export interface CNPGHAInstance { + pod: string + podUID: string + role: 'primary' | 'replica' | 'unknown' + ready: boolean + node?: string + zone?: string + qosClass?: string + image?: string + imageMatches?: boolean + podCreatedAt?: string + postgresStartedAt?: string + restartCount: number +} + +export interface CNPGHAQuorum { + enabled: boolean + enabledBy?: 'spec' | 'annotation' + method?: string + number?: number + dataDurability?: string + object: CNPGHASource + status?: { method?: string; standbyNames: string[]; standbyNumber: number; primary?: string } + n?: number + w?: number + r?: number + promotable?: string[] + holds?: boolean +} + +export interface CNPGHAPDB { + name: string + role: 'primary' | 'replicas' | 'other' + minAvailable?: string + maxUnavailable?: string + expectedPods: number + currentHealthy: number + desiredHealthy: number + disruptionsAllowed: number + observed: boolean +} + +export interface CNPGHALease extends CNPGHASource { + namespace?: string + name?: string + holder?: string + renewTime?: string + durationSeconds?: number + expired?: boolean + controlledByCluster?: boolean +} + +export interface CNPGHAJob { + name: string + role?: string + instance?: string + phase: 'running' | 'active' | 'succeeded' | 'failed' | 'pending' + reason?: string + startTime?: string + completionTime?: string +} + +export interface CNPGHACertificate { + secret: string + purposes?: string[] + raw: string + expiresAt?: string + renewal: 'operator' | 'user' + metadata?: CNPGHASource + certManager?: { certificate: string; issuer?: string; issuerKind?: string } +} + +export interface CNPGMaintenanceFacts { + declared: boolean + inProgress: boolean + reusePVC: boolean +} + +export interface CNPGClusterHA { + cluster: { namespace: string; name: string; uid: string } + sampledAt: string + desiredImage?: string + declaredInstances?: number + expectedInstances?: string[] + instances: CNPGHAInstance[] + pods: CNPGHASource + nodes: CNPGHASource + quorum: CNPGHAQuorum + pdbs: CNPGHASource & { enabled: boolean; items: CNPGHAPDB[] } + primaryLease: CNPGHALease + operatorLease: CNPGHALease + jobs: CNPGHASource & { items: CNPGHAJob[] } + rwEndpoints: CNPGHASource & { service: string; pods: string[] } + certificates: CNPGHACertificate[] + maintenance: CNPGMaintenanceFacts +} + +/** + * Per-instance facts read live from each instance manager (/pg/status). Only + * callers who can read them pass them in: they never come from the cached + * Cluster object. + */ +export interface CNPGInstanceLive { + pod: string + /** ok, or why the instance's own report is missing. */ + state: string + pendingRestart?: boolean + pendingRestartForDecrease?: boolean + /** The report did not finish its reads, so pendingRestart is not established. */ + incomplete?: boolean + /** Why the report is missing or incomplete, in words. */ + reason?: string + roleDetail?: 'primary' | 'pgRewind' | 'replayPaused' | 'streaming' | 'fileBased' + instanceManagerVersion?: string + timeline?: number +} + +export const CNPG_ROLE_DETAIL_TEXT: Record, string> = { + primary: 'primary', + pgRewind: 'pg_rewind running', + replayPaused: 'replay paused', + streaming: 'streaming standby', + fileBased: 'file-based standby (no WAL receiver)', +} + +export function cnpgHASourceText(src: CNPGHASource | undefined, what: string): string { + if (!src) return `${what}: unknown` + switch (src.state) { + case 'ok': + return '' + case 'denied': + return `No access to ${what}${src.grant ? ` (needs ${formatGrant(src.grant)})` : ''}` + case 'notInstalled': + return src.reason ?? `${what}: not available in this CloudNativePG version` + case 'notFound': + return src.reason ?? `No ${what}` + case 'unavailable': + return src.reason ?? `${what}: unavailable` + default: + return src.reason ? `${what} could not be read: ${src.reason}` : `${what} could not be read` + } +} + +// --------------------------------------------------------------------------- +// Failure domains + +export interface CNPGZoneSpread { + /** false when Nodes (or Pods) could not be read: zones are unknown. */ + known: boolean + zones: { zone: string; pods: string[] }[] + /** Instances whose Node carries no zone label. */ + unlabelled: string[] + primaryZone?: string + /** true when every labelled instance shares one zone (and there is more than one instance). */ + singleZone: boolean + nodes: { node: string; pods: string[] }[] + /** true when two or more instances share one Node. */ + sharedNode: boolean +} + +export function cnpgZoneSpread(ha: CNPGClusterHA | undefined): CNPGZoneSpread { + const empty: CNPGZoneSpread = { known: false, zones: [], unlabelled: [], singleZone: false, nodes: [], sharedNode: false } + if (!ha || ha.pods.state !== 'ok') return empty + const byNode = new Map() + for (const i of ha.instances) { + if (!i.node) continue + byNode.set(i.node, [...(byNode.get(i.node) ?? []), i.pod]) + } + const nodes = [...byNode.entries()].map(([node, pods]) => ({ node, pods })).sort((a, b) => a.node.localeCompare(b.node)) + const sharedNode = nodes.some((n) => n.pods.length > 1) + if (ha.nodes.state !== 'ok') return { ...empty, nodes, sharedNode } + const byZone = new Map() + const unlabelled: string[] = [] + let primaryZone: string | undefined + for (const i of ha.instances) { + if (!i.zone) { + unlabelled.push(i.pod) + continue + } + byZone.set(i.zone, [...(byZone.get(i.zone) ?? []), i.pod]) + if (i.role === 'primary') primaryZone = i.zone + } + const zones = [...byZone.entries()].map(([zone, pods]) => ({ zone, pods })).sort((a, b) => a.zone.localeCompare(b.zone)) + return { + known: true, + zones, + unlabelled, + primaryZone, + singleZone: zones.length === 1 && ha.instances.length > 1 && unlabelled.length === 0, + nodes, + sharedNode, + } +} + +// --------------------------------------------------------------------------- +// Quorum, PDB, images, certificates + +export function cnpgQuorumFact(q: CNPGHAQuorum | undefined): Fact { + if (!q) return { text: 'Unknown', tone: 'unknown' } + if (!q.enabled) { + if (q.number !== undefined || q.method) { + return { text: `Synchronous ${q.method ?? ''} ${q.number ?? ''}`.replace(/\s+/g, ' ').trim() + ' · quorum failover off', tone: 'neutral', source: 'Cluster spec.postgresql.synchronous' } + } + return { text: 'Off (asynchronous replication)', tone: 'neutral', source: 'Cluster spec' } + } + const unread = cnpgHASourceText(q.object, 'FailoverQuorum') + if (q.object.state !== 'ok') return { text: `Quorum failover on · ${unread}`, tone: 'unknown' } + if (q.n === undefined || q.w === undefined) { + return { + text: 'Quorum failover on · no synchronous configuration recorded: a failover would wait', + tone: 'degraded', + source: 'FailoverQuorum status (reset while PostgreSQL configuration changes)', + } + } + const base = `W ${q.w} of N ${q.n} potentially synchronous` + if (q.r === undefined || q.holds === undefined) { + return { text: `${base} · promotable replicas unknown`, tone: 'unknown', source: 'FailoverQuorum status; Pods not readable' } + } + return { + text: `${base} · R ${q.r} promotable · R + W ${q.holds ? '>' : '≤'} N`, + tone: q.holds ? 'healthy' : 'degraded', + source: 'FailoverQuorum status (the recorded configuration, not the operator’s decision); R from ready standby Pods', + } +} + +export function cnpgPDBFact(pdbs: CNPGClusterHA['pdbs'] | undefined): Fact { + if (!pdbs) return { text: 'Unknown', tone: 'unknown' } + if (pdbs.state !== 'ok') return { text: cnpgHASourceText(pdbs, 'PodDisruptionBudgets'), tone: 'unknown' } + if (pdbs.items.length === 0) { + return pdbs.enabled + ? { text: 'None found although spec.enablePDB is on', tone: 'degraded', source: 'PodDisruptionBudgets owned by the Cluster' } + : { text: 'Disabled (spec.enablePDB: false): node drains are not held back', tone: 'neutral', source: 'Cluster spec' } + } + const parts = pdbs.items.map((p) => `${p.role === 'primary' ? 'primary' : p.role === 'replicas' ? 'standbys' : p.name}: ${p.disruptionsAllowed} disruption${p.disruptionsAllowed === 1 ? '' : 's'} allowed (${p.currentHealthy}/${p.expectedPods} healthy)`) + const stale = pdbs.items.some((p) => !p.observed) + return { + text: parts.join(' · '), + tone: stale ? 'unknown' : 'neutral', + source: stale ? 'PodDisruptionBudget status (not yet updated for the latest spec)' : 'PodDisruptionBudget status', + } +} + +export function cnpgImageDrift(ha: CNPGClusterHA | undefined): { known: boolean; drifted: CNPGHAInstance[] } { + if (!ha || ha.pods.state !== 'ok' || !ha.desiredImage || ha.instances.length === 0) return { known: false, drifted: [] } + return { known: ha.instances.every((i) => i.imageMatches !== undefined), drifted: ha.instances.filter((i) => i.imageMatches === false) } +} + +export interface CNPGCertificateView extends CNPGHACertificate { + /** Whole days until expiry; negative when expired; undefined when the expiry did not parse. */ + daysLeft?: number + tone: HealthLevel +} + +/** + * The same thresholds as the Issues engine: a certificate its owner renews is + * flagged from 30 days; the operator renews its own, so those only matter once + * renewal is overdue. + */ +export function cnpgCertificateViews(certs: CNPGHACertificate[] | undefined, now = Date.now()): CNPGCertificateView[] { + return (certs ?? []).map((c) => { + if (!c.expiresAt) return { ...c, tone: 'unknown' as HealthLevel } + const ms = Date.parse(c.expiresAt) - now + const daysLeft = Math.floor(ms / 86_400_000) + let tone: HealthLevel = 'healthy' + if (ms <= 0) tone = 'unhealthy' + else if (c.renewal === 'user' && ms < 7 * 86_400_000) tone = 'unhealthy' + else if (c.renewal === 'user' && ms < 30 * 86_400_000) tone = 'degraded' + else if (c.renewal === 'operator' && ms < 86_400_000) tone = 'unhealthy' + return { ...c, daysLeft, tone } + }) +} + +/** + * "HA and instances" in one line: what is wrong when something is, otherwise + * the readiness and placement facts that are known. Unknown facts never read + * as fine; they are left out of a calm summary rather than claimed. + */ +export function cnpgHASummary( + ha: CNPGClusterHA | undefined, + live: CNPGInstanceLive[] | undefined, + primaryConflict?: { status: string; labelled: string }, +): FoldSummary { + if (!ha) return { text: 'Not read', attention: false } + const issues: string[] = [] + const calm: string[] = [] + if (ha.pods.state === 'ok') { + const ready = ha.instances.filter((i) => i.ready).length + const missing = [...new Set(ha.expectedInstances ?? [])].filter((name) => !ha.instances.some((i) => i.pod === name)) + if (ha.declaredInstances !== undefined && ready < ha.declaredInstances) { + issues.push(`${ready} of ${ha.declaredInstances} declared instances ready${missing.length ? `; no instance Pod observed for ${missing.join(', ')}` : ''}`) + } else if (ha.declaredInstances !== undefined && ready > ha.declaredInstances) { + issues.push(`${ready} observed instances ready; ${ha.declaredInstances} instances declared`) + } else if (ha.instances.length === 0) issues.push('no instance Pods') + else if (ready < ha.instances.length) issues.push(`${ha.instances.length - ready} of ${ha.instances.length} instances not ready`) + else calm.push(ha.declaredInstances !== undefined ? `${ready} of ${ha.declaredInstances} declared instances ready` : `${ready}/${ha.instances.length} observed instances ready`) + } + if (primaryConflict) issues.push('primary labels disagree') + const spread = cnpgZoneSpread(ha) + if (spread.singleZone) issues.push('every instance in one zone') + if (spread.sharedNode) issues.push('instances share a Node') + if (spread.known && !spread.singleZone && spread.zones.length > 1) calm.push(`${spread.zones.length} zones`) + const pending = cnpgPendingRestart(live) + if (pending.pods.length > 0) issues.push(`restart pending on ${pending.pods.join(', ')}`) + const drift = cnpgImageDrift(ha) + if (drift.drifted.length > 0) issues.push(`${drift.drifted.length === 1 ? 'an instance Pod has' : `${drift.drifted.length} instance Pods have`} a different image`) + else if (drift.known && ha.instances.length > 0 && ha.instances.every((i) => i.imageMatches === true)) calm.push('images match') + const quorum = cnpgQuorumFact(ha.quorum) + if (quorum.tone === 'degraded' || quorum.tone === 'unhealthy') issues.push('failover quorum does not hold') + const pdb = cnpgPDBFact(ha.pdbs) + if (pdb.tone === 'degraded' || pdb.tone === 'unhealthy') issues.push('disruption budgets missing') + for (const [lease, what] of [[ha.primaryLease, 'primary lease'], [ha.operatorLease, 'operator lease']] as const) { + if (lease.state === 'ok' && lease.expired) issues.push(`${what} expired`) + } + if (ha.primaryLease.state === 'ok' && ha.primaryLease.controlledByCluster === false) issues.push('primary lease not owned by this Cluster') + const failedJobs = ha.jobs.state === 'ok' ? ha.jobs.items.filter((j) => j.phase === 'failed').length : 0 + if (failedJobs > 0) issues.push(`${failedJobs} failed ${failedJobs === 1 ? 'Job' : 'Jobs'}`) + // A fact that could not be read is named, never folded into a calm line. + const unread: string[] = [] + const check = (src: CNPGHASource | undefined, label: string) => { + if (src && (src.state === 'denied' || src.state === 'unavailable' || src.state === 'error')) unread.push(label) + } + check(ha.pods, 'Pods') + check(ha.nodes, 'zones') + if (ha.quorum.enabled) check(ha.quorum.object, 'quorum') + check(ha.pdbs, 'disruption budgets') + check(ha.primaryLease, 'primary lease') + check(ha.operatorLease, 'operator lease') + check(ha.jobs, 'Jobs') + if (!pending.known && !(ha.pods.state === 'ok' && ha.instances.length === 0)) unread.push('pending restarts') + const notRead = unread.length > 0 ? `not read: ${unread.join(', ')}` : '' + if (issues.length > 0) return { text: [...issues, notRead].filter(Boolean).join(' · '), attention: true } + if (calm.length === 0) return { text: notRead ? notRead[0].toUpperCase() + notRead.slice(1) : 'Nothing reported out of line', attention: false } + return { text: [...calm, notRead].filter(Boolean).join(' · '), attention: false } +} + +/** Certificates in one line: the nearest expiry and who renews them. */ +export function cnpgCertificatesSummary(certs: CNPGHACertificate[] | undefined, now = Date.now()): FoldSummary { + const views = cnpgCertificateViews(certs, now) + if (views.length === 0) return { text: 'No expiry reported by the operator', attention: false } + const dated = views.filter((c) => Number.isFinite(c.daysLeft)).sort((a, b) => (a.daysLeft ?? 0) - (b.daysLeft ?? 0)) + const unreadable = views.length - dated.length + // An unreadable expiry could be past already, so it opens the section too. + const attention = unreadable > 0 || views.some((c) => c.tone === 'degraded' || c.tone === 'unhealthy') + const nearest = dated[0] + const when = [ + !nearest ? '' : nearest.daysLeft! < 0 ? `${nearest.secret} expired` : `nearest reported expiry in ${nearest.daysLeft} d (${nearest.secret})`, + unreadable > 0 ? `${unreadable} expiry unreadable` : '', + ] + .filter(Boolean) + .join(' · ') + const renewers = new Set(views.map((c) => c.renewal)) + const who = renewers.size > 1 ? 'some renewed by you' : renewers.has('user') ? 'you renew them' : 'CloudNativePG renews them' + return { text: `${views.length} ${views.length === 1 ? 'certificate' : 'certificates'} · ${when} · ${who}`, attention } +} + +function cnpgLiveRead(l: CNPGInstanceLive): boolean { + return (l.state === 'ok' || l.state === 'partial') && !l.incomplete +} + +export function cnpgPendingRestart(live: CNPGInstanceLive[] | undefined): { known: boolean; pods: string[]; forDecrease: boolean } { + const read = (live ?? []).filter(cnpgLiveRead) + if (read.length === 0) return { known: false, pods: [], forDecrease: false } + const pending = read.filter((l) => l.pendingRestart) + return { known: read.length === (live ?? []).length, pods: pending.map((l) => l.pod), forDecrease: pending.some((l) => l.pendingRestartForDecrease) } +} + +/** + * Why instance-manager facts are not established: the host's reason when + * there is no runtime read at all (`unavailable`, e.g. the missing grant), + * otherwise each instance that did not report and why. + */ +export function cnpgLiveGap(live: CNPGInstanceLive[] | undefined, unavailable?: string): string { + if (!live) return unavailable ?? 'needs each instance manager’s status (get pods/proxy)' + if (live.length === 0) return 'no instance was read' + const unread = live.filter((l) => !cnpgLiveRead(l)) + if (unread.length === 0) return '' + return unread + .map((l) => { + const why = l.reason ?? (l.state === 'denied' ? 'no access' : l.state) + return l.incomplete ? `${l.pod} reported incompletely (${why})` : `${l.pod} did not report (${why})` + }) + .join('; ') +} + +// --------------------------------------------------------------------------- +// Header dimensions + +export type CNPGDimensionId = 'serving' | 'replication' | 'protection' | 'storage' + +export interface CNPGDimension { + id: CNPGDimensionId + label: string + /** unknown = unassessed: its source is not available. */ + tone: HealthLevel + text: string + source: string +} + +export interface CNPGReplicationLive { + /** Standbys the primary reports as streaming. */ + streaming: number + /** Standby Pods the runtime read saw; the verdict compares against spec.instances − 1, not this. */ + standbys: number + maxReplayLagSeconds?: number +} + +export function cnpgDimensions({ + row, + ha, + replication, + replicationGap, + storage, +}: { + row: CNPGFleetRow + ha?: CNPGClusterHA + /** From the primary's pg_stat_replication; absent when runtime data is not readable. */ + replication?: CNPGReplicationLive + /** Why `replication` is absent (e.g. "needs get pods/proxy in db"). */ + replicationGap?: string + /** Supplied by the host once storage is assessed; unassessed otherwise. */ + storage?: CNPGDimension +}): CNPGDimension[] { + return [ + servingDimension(row, ha), + replicationDimension(row, replication, replicationGap, ha), + storage ?? storageDimension(row), + protectionDimension(row), + ] +} + +function storageDimension(row: CNPGFleetRow): CNPGDimension { + return withSlotRetention(row, volumeDimension(row)) +} + +function volumeDimension(row: CNPGFleetRow): CNPGDimension { + const base = { id: 'storage' as const, label: 'Storage' } + const disk = row.disk + if (!disk) return { ...base, tone: 'unknown', text: 'Reading…', source: 'Reading volume usage' } + if (disk.tone === 'unknown') return { ...base, tone: 'unknown', text: disk.source === CNPG_PROMETHEUS_NOT_CONNECTED ? disk.text : [disk.text, disk.source].filter(Boolean).join(': '), source: disk.source === CNPG_PROMETHEUS_NOT_CONNECTED ? '' : disk.detail ?? '' } + return { ...base, tone: disk.tone, text: disk.text, source: disk.source ?? 'Fullest volume' } +} + +// WAL an inactive slot pins is a storage concern even while volume usage is +// unmeasured; the chip says so instead of reading "unassessed". +function withSlotRetention(row: CNPGFleetRow, dim: CNPGDimension): CNPGDimension { + const slot = row.problems.find((p) => p.reason === 'CNPGInactiveSlot') + if (!slot) return dim + const held = 'WAL held by an inactive slot' + return { + ...dim, + tone: worseTone(dim.tone, 'degraded'), + text: dim.tone === 'unknown' ? held : `${dim.text} · ${held}`, + source: dim.tone === 'unknown' ? `${slot.title}. Volume usage: ${[dim.text, dim.source].filter(Boolean).join(' · ')}` : `${slot.title}. ${dim.source ?? ''}`.trim(), + } +} + +function servingDimension(row: CNPGFleetRow, ha?: CNPGClusterHA): CNPGDimension { + const base = { id: 'serving' as const, label: 'Serving' } + if (row.hibernated) return { ...base, tone: 'neutral', text: 'hibernated', source: 'cnpg.io/hibernation annotation' } + if (ha?.rwEndpoints.state === 'ok' && ha.rwEndpoints.pods.length === 0) return { ...base, tone: 'unhealthy', text: 'not serving: no ready read-write endpoint', source: `EndpointSlices of Service ${ha.rwEndpoints.service}` } + const primaryName = row.cluster?.status?.currentPrimary as string | undefined + const primary = row.pods.find((p) => p.name === primaryName) + if (!primaryName) return { ...base, tone: 'unknown', text: 'unassessed', source: 'No current primary reported' } + if (!primary || primary.ready === null) return { ...base, tone: 'unknown', text: 'unassessed', source: 'Instance Pods are not readable' } + if (!primary.ready) return { ...base, tone: 'unhealthy', text: 'primary not ready', source: `Pod ${primaryName} readiness` } + if (ha?.rwEndpoints.state === 'ok') { + if (!ha.rwEndpoints.pods.includes(primaryName)) { + return { + ...base, + tone: 'unhealthy', + text: ha.rwEndpoints.pods.length === 0 ? 'no read-write endpoint' : 'read-write endpoint not on the primary', + source: `EndpointSlices of Service ${ha.rwEndpoints.service}`, + } + } + return { ...base, tone: 'healthy', text: 'primary ready', source: `Pod ${primaryName} ready and behind Service ${ha.rwEndpoints.service}` } + } + return { ...base, tone: 'healthy', text: 'primary ready', source: `Pod ${primaryName} readiness (Service endpoints not readable)` } +} + +// A standby measured far behind for the whole window outranks what the live +// read shows: the chip must not read calmer than the finding below it. +function replicationDimension(row: CNPGFleetRow, live?: CNPGReplicationLive, gap?: string, ha?: CNPGClusterHA): CNPGDimension { + let dim = withStandbyGaps(row, liveReplicationDimension(row, live, gap)) + if (!row.hibernated && !row.replicaCluster && (row.instances.desired ?? 0) > 1) { + const primary = row.cluster?.status?.currentPrimary + const pods = ha?.pods.state === 'ok' ? ha.instances.map((i) => ({ name: i.pod, ready: i.ready })) : row.podReadiness ? row.pods : undefined + if (pods && primary) { + const expected = ha?.pods.state === 'ok' ? ha.expectedInstances ?? row.cluster?.status?.instanceNames ?? [] : row.cluster?.status?.instanceNames ?? [] + const missing = expected.filter((name: string) => name !== primary && !pods.some((p) => p.name === name)) + if (!live && missing.length > 0) { + const otherGaps = row.problems.filter((p) => p.reason === 'CNPGStandbyNotReceiving' && !missing.includes(p.subject.name)) + const measuredGaps = withStandbyGaps({ ...row, problems: otherGaps }, liveReplicationDimension(row, live, gap)) + dim = { + ...dim, tone: worseTone(dim.tone, 'degraded'), + text: `Expected standby${missing.length === 1 ? '' : 's'} ${missing.join(', ')} ${missing.length === 1 ? 'is' : 'are'} not running${otherGaps.length > 0 ? ` · ${measuredGaps.text}` : ''}`, + source: `${otherGaps.length > 0 ? `${measuredGaps.source}; ` : ''}Instance Pods; streaming not measured: ${gap ?? 'not read'}`, + } + } + if (pods.some((p) => p.name === primary && p.ready === true) && !pods.some((p) => p.name !== primary && p.ready === true)) dim = { + ...dim, tone: worseTone(dim.tone, 'degraded'), source: `${dim.source}. No ready standby to fail over to`, + } + } + } + const sustained = row.problems.find((p) => p.reason === 'CNPGSustainedLag') + if (!sustained) return dim + const tone: HealthLevel = sustained.severity === 'critical' ? 'unhealthy' : 'degraded' + return { + ...dim, + tone: worseTone(dim.tone, tone), + text: dim.tone === 'unknown' ? 'sustained lag' : `${dim.text} · sustained lag`, + source: sustained.title, + } +} + +// A standby another source saw receiving nothing (Prometheus in the fleet, the +// instance manager here) keeps the chip from reading calm or unassessed. +function withStandbyGaps(row: CNPGFleetRow, dim: CNPGDimension): CNPGDimension { + const gaps = row.problems.filter((p) => p.reason === 'CNPGStandbyNotReceiving') + if (gaps.length === 0) return dim + const pods = gaps.map((p) => p.subject.name).join(', ') + const tone: HealthLevel = gaps.some((p) => p.severity === 'critical') ? 'unhealthy' : 'degraded' + return { + ...dim, + tone: worseTone(dim.tone, tone), + text: dim.tone === 'unknown' ? `${pods} not receiving WAL` : `${dim.text} · ${pods} not receiving WAL`, + source: gaps.map((p) => p.title).join('; '), + } +} + +function liveReplicationDimension(row: CNPGFleetRow, live?: CNPGReplicationLive, gap?: string): CNPGDimension { + const base = { id: 'replication' as const, label: 'Replication' } + if (row.hibernated) return { ...base, tone: 'neutral', text: 'hibernated', source: 'cnpg.io/hibernation annotation' } + if (row.instances.desired === 1) return { ...base, tone: 'degraded', text: 'no standby', source: 'spec.instances is 1: there is no failover target' } + if (!live) return { ...base, tone: 'unknown', text: 'unassessed', source: `Needs the primary’s pg_stat_replication: ${gap ?? 'read through get pods/proxy'}` } + // Standbys whose Pods are gone are missing from the runtime read too, so + // the denominator is what the Cluster asks for, never what is running. + const desired = row.instances.desired + if (desired === null) { + return { ...base, tone: 'unknown', text: `${live.streaming} streaming`, source: 'spec.instances is not reported, so the expected standbys are unknown' } + } + const expected = Math.max(0, desired - 1) + const source = `Primary’s pg_stat_replication against spec.instances ${desired}` + const lag = live.maxReplayLagSeconds + const tone = cnpgReplicationTone(live.streaming, expected, lag) + // pg_stat_replication's replay_lag is the recent replay delay, not WAL still + // to replay; it can stay high after a standby caught up, so it is not "behind". + const lagText = lag !== undefined && cnpgLagTone(lag) !== 'healthy' ? ` · replay delay ${cnpgFormatLag(lag)}` : '' + if (live.streaming < expected) { + return { ...base, tone, text: `${live.streaming} of ${expected} expected standbys streaming${lagText}`, source } + } + if (lagText) return { ...base, tone, text: `${live.streaming} of ${expected} streaming${lagText}`, source } + return { ...base, tone: 'healthy', text: `${live.streaming} of ${expected} standbys streaming`, source } +} + +function protectionDimension(row: CNPGFleetRow): CNPGDimension { + const base = { id: 'protection' as const, label: 'Backups' } + const p = row.protection + if (p.walArchiving.tone === 'unhealthy') return { ...base, tone: 'unhealthy', text: 'WAL archiving failing', source: 'ContinuousArchiving condition' } + const blockedSchedule = row.problems.find((problem) => problem.reason === 'CNPGScheduleDestinationMissing') + if (blockedSchedule) return { ...base, tone: 'degraded', text: blockedSchedule.title, source: blockedSchedule.origin?.detail ?? 'ScheduledBackup method against Cluster spec' } + if (p.destination.method === 'none') return { ...base, tone: 'degraded', text: 'no backup destination', source: 'Cluster spec' } + if (p.lastSuccessfulBackup.tone === 'unhealthy' || p.lastSuccessfulBackup.tone === 'degraded') { + const text = !p.lastSuccessfulBackup.at && p.walArchiving.tone === 'healthy' + ? `CNPG reports archiving · ${p.lastSuccessfulBackup.text.toLowerCase()}` + : p.lastSuccessfulBackup.text + return { ...base, tone: p.lastSuccessfulBackup.tone, text, source: p.lastSuccessfulBackup.source ?? 'Backups' } + } + if (p.lastSuccessfulBackup.tone === 'unknown') return { ...base, tone: 'unknown', text: 'unassessed', source: p.lastSuccessfulBackup.source ?? p.lastSuccessfulBackup.text } + if (p.walArchiving.tone === 'unknown') return { ...base, tone: 'unknown', text: 'unassessed', source: 'WAL archiving not reported' } + const last = p.lastSuccessfulBackup.at ? ` · last backup ${formatAge(p.lastSuccessfulBackup.at)} ago` : '' + if (p.walArchiving.tone === 'neutral') return { ...base, tone: 'healthy', text: `Backup completed${last}`, source: p.lastSuccessfulBackup.source ?? 'Backups' } + return { ...base, tone: 'healthy', text: `CNPG reports archiving${last}`, source: `ContinuousArchiving condition${p.lastSuccessfulBackup.source ? ` · last backup: ${p.lastSuccessfulBackup.source}` : ''}` } +} + +/** + * The Pod a Lease holder names. controller-runtime's leader election records + * "_"; a CNPG primary Lease records the Pod name alone. + */ +export function cnpgLeaseHolderPod(holder: string): string { + const i = holder.indexOf('_') + return i > 0 ? holder.slice(0, i) : holder +} diff --git a/packages/k8s-ui/src/components/cnpg/index.ts b/packages/k8s-ui/src/components/cnpg/index.ts index 08bd46e1fb..7a12978822 100644 --- a/packages/k8s-ui/src/components/cnpg/index.ts +++ b/packages/k8s-ui/src/components/cnpg/index.ts @@ -1,4 +1,7 @@ export * from './workspace' +export * from './databaseRole' +export * from './ha' +export * from './CNPGClusterHASection' export * from './primitives' export * from './CNPGClusterSummary' export * from './CNPGBackupSummary' @@ -6,9 +9,24 @@ export * from './CNPGObjectStoreSummary' export * from './CNPGDeclarativeSummary' export * from './CNPGPoolerSummary' export * from './CNPGImageCatalogSummary' +export * from './pooler' +export * from './connect' +export * from './CNPGConnectSection' +export * from './schedule' +export * from './logicalReplication' +export * from './CNPGLogicalPath' export { + backupsForScheduledBackup, + cnpgScheduleDestinationBlocker, + objectStoreForBackup, + cnpgBackupMatchesCluster, + cnpgArchiveMatchesCluster, + type CNPGArchiveSource, inferredObjectStoreHealth, + relationUnavailable, usersOfObjectStore, type CNPGObjectStoreHealth, type CNPGObjectStoreUser, } from './relations' + +export * from './backupRuns' diff --git a/packages/k8s-ui/src/components/cnpg/logicalReplication.test.ts b/packages/k8s-ui/src/components/cnpg/logicalReplication.test.ts new file mode 100644 index 0000000000..7debb9cf5a --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/logicalReplication.test.ts @@ -0,0 +1,121 @@ +import { describe, expect, it } from 'vitest' +import { cnpgLogicalLocation, cnpgLogicalPaths, cnpgLogicalSlotFact, cnpgResolvePublisher } from './logicalReplication' + +const G = 'postgresql.cnpg.io/v1' +const cluster = (name: string, ns: string, spec: any = {}, status: any = {}) => ({ apiVersion: G, kind: 'Cluster', metadata: { name, namespace: ns }, spec: { instances: 3, ...spec }, status }) +const sub = (spec: any, status: any = {}) => ({ apiVersion: G, kind: 'Subscription', metadata: { name: 'orders-sub', namespace: 'dst' }, spec: { cluster: { name: 'dst' }, name: 'orders_sub', dbname: 'app', publicationName: 'orders_pub', externalClusterName: 'src', ...spec }, status }) +const pub = { apiVersion: G, kind: 'Publication', metadata: { name: 'orders-pub', namespace: 'src' }, spec: { cluster: { name: 'src' }, name: 'orders_pub', dbname: 'app', target: { allTables: true } }, status: { applied: true } } +const subscriber = (host: string) => cluster('dst', 'dst', { externalClusters: [{ name: 'src', connectionParameters: { host, dbname: 'app', user: 'app' } }] }) +const synced = (major: number) => cluster('src', 'src', { replicationSlots: { highAvailability: { synchronizeLogicalDecoding: true } } }, { pgDataImageInfo: { majorVersion: major } }) + +describe('cnpgResolvePublisher', () => { + const clusters = [cluster('src', 'src'), cluster('dst', 'dst')] + it('resolves -rw/-ro/-r Service hosts in any spelling to the visible Cluster', () => { + for (const host of ['src-rw.src.svc', 'src-rw.src.svc.cluster.local', 'src-ro.src', 'SRC-R.src.svc']) { + const p = cnpgResolvePublisher(host, 'dst', clusters, []) + expect(p.kind === 'cluster' && p.name).toBe('src') + } + }) + it('resolves through a Pooler Service, and a bare name in the subscriber namespace', () => { + const p = cnpgResolvePublisher('src-pooler.src.svc', 'dst', clusters, [{ metadata: { name: 'src-pooler', namespace: 'src' }, spec: { cluster: { name: 'src' } } }]) + expect(p.kind === 'cluster' && p.via).toBe('Pooler src-pooler') + expect(cnpgResolvePublisher('dst-rw', 'dst', clusters, []).kind).toBe('cluster') + }) + it('never guesses a cluster from an external host', () => { + expect(cnpgResolvePublisher('src-rw.example.com', 'dst', clusters, []).kind).toBe('external') + expect(cnpgResolvePublisher('other-rw.src.svc', 'dst', clusters, []).kind).toBe('external') + expect(cnpgResolvePublisher(undefined, 'dst', clusters, []).kind).toBe('external') + }) +}) + +describe('cnpgLogicalPaths', () => { + it('walks subscription → external cluster → publisher → publication object → slot', () => { + const [p] = cnpgLogicalPaths([sub({})], [synced(17), subscriber('src-rw.src.svc')], [pub], []) + expect(p.publisher.kind === 'cluster' && `${p.publisher.namespace}/${p.publisher.name}`).toBe('src/src') + expect(p.publication.object?.name).toBe('orders-pub') + expect(p.publication.dbname).toBe('app') + expect(p.slot.name).toBe('orders_sub') + }) + it('says when the external cluster is not declared or the publication is not a CNPG object', () => { + const [missing] = cnpgLogicalPaths([sub({ externalClusterName: 'nope' })], [synced(17), subscriber('src-rw.src.svc')], [pub], []) + expect(missing.publisher.kind).toBe('external') + expect(missing.externalCluster.declared).toBe(false) + const [sqlOnly] = cnpgLogicalPaths([sub({ publicationName: 'made_in_sql' })], [synced(17), subscriber('src-rw.src.svc')], [pub], []) + expect(sqlOnly.publication.object).toBeUndefined() + }) + it('says a Publication is unknown when the publisher namespace\'s Publications are unreadable', () => { + const [p] = cnpgLogicalPaths([sub({})], [synced(17), subscriber('src-rw.src.svc')], [], [], (ns) => (ns === 'src' ? 'No access to Publications' : null)) + expect(p.publication.object).toBeUndefined() + expect(p.publication.unavailable).toBe('No access to Publications') + const [readable] = cnpgLogicalPaths([sub({})], [synced(17), subscriber('src-rw.src.svc')], [], [], () => null) + expect(readable.publication.unavailable).toBeUndefined() + }) + it('names the slot from slot_name, and none for slot_name = NONE', () => { + expect(cnpgLogicalPaths([sub({ parameters: { slot_name: 'custom' } })], [], [], [])[0].slot.name).toBe('custom') + expect(cnpgLogicalPaths([sub({ parameters: { slot_name: 'NONE' } })], [], [], [])[0].slot.name).toBeUndefined() + }) +}) + +describe('slot failover', () => { + const failoverOf = (publisher: any, params?: any) => + cnpgLogicalPaths([sub(params ? { parameters: params } : {})], [publisher, subscriber('src-rw.src.svc')], [], [])[0].failover + it('is lost when the publisher does not synchronize logical slots', () => { + const f = failoverOf(cluster('src', 'src')) + expect(f.tone).toBe('degraded') + expect(f.text).toContain('synchronizeLogicalDecoding is off') + }) + it('on PostgreSQL 17 needs the subscription to request failover', () => { + expect(failoverOf(synced(17)).tone).toBe('degraded') + expect(failoverOf(synced(17), { failover: 'true' }).tone).toBe('healthy') + }) + it('is lost when HA slots are disabled, whatever synchronizeLogicalDecoding says', () => { + const off = cluster('src', 'src', { replicationSlots: { highAvailability: { enabled: false, synchronizeLogicalDecoding: true } } }, { pgDataImageInfo: { majorVersion: 17 } }) + const f = failoverOf(off, { failover: 'true' }) + expect(f.tone).toBe('degraded') + expect(f.text).toContain('HA replication slots are disabled') + }) + it('is unknown before 17 (pg_failover_slots), without a major, or outside Radar', () => { + expect(failoverOf(synced(16)).tone).toBe('unknown') + expect(failoverOf(cluster('src', 'src', { replicationSlots: { highAvailability: { synchronizeLogicalDecoding: true } } })).tone).toBe('unknown') + expect(cnpgLogicalPaths([sub({})], [subscriber('db.example.com')], [], [])[0].failover.tone).toBe('unknown') + }) +}) + +describe('cnpgLogicalSlotFact', () => { + const [path] = cnpgLogicalPaths([sub({}, { applied: true })], [synced(17), subscriber('src-rw.src.svc')], [pub], []) + it('never reads a denied or unread runtime as a missing slot', () => { + expect(cnpgLogicalSlotFact(path, { state: 'denied' }).tone).toBe('unknown') + expect(cnpgLogicalSlotFact(path, { state: 'unavailable', reason: 'unreachable' }).text).toContain('not read (unreachable)') + }) + it('reports the observed slot, and a missing one only from a readable report', () => { + const ok = cnpgLogicalSlotFact(path, { state: 'ok', slots: [{ name: 'orders_sub', type: 'logical', active: true, retainedBytes: 2048, walStatus: 'reserved' }] }) + expect(ok.text).toBe('Slot orders_sub · logical · active · retains 2.0 KiB of WAL · WAL reserved') + expect(ok.tone).toBe('healthy') + expect(cnpgLogicalSlotFact(path, { state: 'ok', slots: [{ name: 'orders_sub', active: false }] }).tone).toBe('degraded') + const capped = cnpgLogicalSlotFact(path, { state: 'partial', reason: '250 replication slots; the first 200 are shown', slots: [] }) + expect(capped.text).toContain('not in the reported slots (report incomplete: 250 replication slots') + expect(capped.tone).toBe('unknown') + const stale = cnpgLogicalSlotFact(path, { state: 'ok', stale: true, slots: [{ name: 'orders_sub', type: 'logical', active: true }] }) + expect(stale.tone).toBe('unknown') + expect(stale.text).toContain('from an earlier read') + const gone = cnpgLogicalSlotFact(path, { state: 'ok', slots: [] }) + expect(gone.text).toContain('not found') + expect(gone.tone).toBe('degraded') + }) +}) + +describe('cnpgLogicalLocation', () => { + it('says an unknown database in words', () => { + expect(cnpgLogicalLocation('upstream', 'app')).toBe('upstream/app') + expect(cnpgLogicalLocation('external cluster upstream', undefined)).toBe('external cluster upstream · database unknown') + expect(cnpgLogicalLocation('x', undefined)).not.toContain('?') + }) +}) + +it('keeps stale Publication and Subscription results pending on the logical path', () => { + const staleSub = { ...sub({}, { applied: true, observedGeneration: 1 }), metadata: { ...sub({}).metadata, generation: 2 } } + const stalePub = { ...pub, metadata: { ...pub.metadata, generation: 2 }, status: { applied: true, observedGeneration: 1 } } + const [path] = cnpgLogicalPaths([staleSub], [synced(17), subscriber('src-rw.src.svc')], [stalePub], []) + expect(path.subscription.applied).toBeNull() + expect(path.publication.object?.applied).toBeNull() +}) diff --git a/packages/k8s-ui/src/components/cnpg/logicalReplication.ts b/packages/k8s-ui/src/components/cnpg/logicalReplication.ts new file mode 100644 index 0000000000..6e961cb451 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/logicalReplication.ts @@ -0,0 +1,252 @@ +import { getCNPGPostgresMajor } from '../resources/resource-utils-cnpg' +import { cnpgRoleState } from './databaseRole' +import { cnpgFormatBytes } from './workspace' +import { type Fact } from '../facts' + +/** Where a Subscription's publisher lives, as far as the subscriber's spec shows. */ +export type CNPGPublisher = + | { kind: 'cluster'; namespace: string; name: string; via: string; cluster: any } + | { kind: 'external'; host?: string; reason: string } + +export interface CNPGLogicalPath { + subscription: { namespace: string; name: string; sqlName?: string; dbname?: string; cluster?: string; applied: boolean | null; message?: string } + externalCluster: { name?: string; declared: boolean; host?: string; dbname?: string } + publisher: CNPGPublisher + publication: { + name?: string + dbname?: string + /** The Publication object on the publisher Cluster declaring it, when one is visible. */ + object?: { namespace: string; name: string; applied: boolean | null } + /** Why Publication objects in the publisher's namespace could not be read; absence then proves nothing. */ + unavailable?: string + } + /** The slot PostgreSQL creates for the subscription: `slot_name`, else the subscription's name. */ + slot: { name?: string; reason?: string } + failover: Fact +} + +const SERVICE_SUFFIXES = ['-rw', '-ro', '-r'] + +function stripHost(host: string): { name: string; namespace?: string } { + const h = host.trim().toLowerCase().replace(/\.$/, '').replace(/\.svc\.cluster\.local$/, '').replace(/\.svc$/, '') + const [name, namespace] = h.split('.') + return { name, namespace } +} + +/** + * Resolves an external cluster's host to a visible CNPG Cluster through its + * -rw / -ro / -r Service or a Pooler's Service. Anything else stays external: + * the host is never guessed to be a cluster it only resembles. + */ +export function cnpgResolvePublisher(host: string | undefined, subscriberNs: string, clusters: any[], poolers: any[]): CNPGPublisher { + if (!host) return { kind: 'external', reason: 'the external cluster names no host' } + const { name, namespace = subscriberNs } = stripHost(host) + if (host.includes('.') && !/\.svc(\.cluster\.local)?\.?$/i.test(host.trim()) && host.split('.').length > 2) { + return { kind: 'external', host, reason: 'the host is not a Kubernetes Service name' } + } + const inNs = (list: any[]) => list.filter((o) => o?.metadata?.namespace === namespace) + for (const suffix of SERVICE_SUFFIXES) { + if (!name.endsWith(suffix)) continue + const base = name.slice(0, -suffix.length) + const c = inNs(clusters).find((x) => x.metadata?.name === base) + if (c) return { kind: 'cluster', namespace, name: base, via: `${name}.${namespace}`, cluster: c } + } + const pooler = inNs(poolers).find((p) => p.metadata?.name === name) + if (pooler?.spec?.cluster?.name) { + const c = inNs(clusters).find((x) => x.metadata?.name === pooler.spec.cluster.name) + if (c) return { kind: 'cluster', namespace, name: c.metadata.name, via: `Pooler ${name}`, cluster: c } + } + return { kind: 'external', host, reason: 'no visible CloudNativePG Cluster serves this host' } +} + +/** + * The namespace a Subscription's publisher host names, so a host can read that + * namespace too before resolving the publisher. Undefined when the subscriber + * or its external cluster is not visible, or the host names none. + */ +export function cnpgSubscriptionHostNamespace(subscription: any, clusters: any[]): string | undefined { + const ns = subscription?.metadata?.namespace + const subscriber = clusters.find((c) => c?.metadata?.namespace === ns && c?.metadata?.name === subscription?.spec?.cluster?.name) + const ext = (subscriber?.spec?.externalClusters ?? []).find((e: any) => e?.name === subscription?.spec?.externalClusterName) + const host = ext?.connectionParameters?.host + return host ? stripHost(host).namespace ?? ns : undefined +} + + +function truthy(v: unknown): boolean | undefined { + if (v === undefined || v === null) return undefined + const s = String(v).trim().toLowerCase() + if (['true', 'on', 'yes', '1'].includes(s)) return true + if (['false', 'off', 'no', '0'].includes(s)) return false + return undefined +} + +const FAILOVER_SOURCE = "Publisher's spec.replicationSlots.highAvailability (enabled and synchronizeLogicalDecoding), its PostgreSQL major and the Subscription's failover parameter; /pg/status does not report a slot's failover flag" + +/** + * Whether the publisher's slot for this subscription is kept on its standbys, + * so a failover of the publisher does not lose it. Declared configuration + * only: CloudNativePG does not report which slots were actually synchronized. + */ +export function cnpgSlotFailover(publisher: CNPGPublisher, subscription: any): Fact { + if (publisher.kind !== 'cluster') { + return { text: 'Unknown: the publisher is not a CloudNativePG Cluster Radar can see', tone: 'unknown', source: FAILOVER_SOURCE } + } + const c = publisher.cluster + if ((c?.spec?.instances ?? 1) <= 1) { + return { text: 'No standby to fail over to: a single-instance publisher', tone: 'neutral', source: FAILOVER_SOURCE } + } + const ha = c?.spec?.replicationSlots?.highAvailability + // GetEnabled defaults to true; the operator enables sync only with both. + if (ha?.synchronizeLogicalDecoding === true && ha?.enabled === false) { + return { + text: 'Lost on failover: synchronizeLogicalDecoding is on, but HA replication slots are disabled, so standbys have no physical slot to synchronize through', + tone: 'degraded', + source: FAILOVER_SOURCE, + } + } + const sync = ha?.synchronizeLogicalDecoding === true + if (!sync) { + return { + text: 'Lost on failover: the publisher does not synchronize logical slots to standbys (synchronizeLogicalDecoding is off)', + tone: 'degraded', + source: FAILOVER_SOURCE, + } + } + const major = getCNPGPostgresMajor(c) + if (major === undefined) { + return { text: 'Unknown: synchronization is on, but the publisher\'s PostgreSQL major is not reported', tone: 'unknown', source: FAILOVER_SOURCE } + } + if (major < 17) { + return { + text: 'Synchronized only if pg_failover_slots is loaded on the publisher (PostgreSQL < 17); Radar cannot tell', + tone: 'unknown', + source: FAILOVER_SOURCE, + } + } + const failover = truthy(subscription?.spec?.parameters?.failover) + if (failover === true) { + return { text: 'Kept on standbys (declared): synchronization is on and the subscription requests failover', tone: 'healthy', source: FAILOVER_SOURCE } + } + return { + text: "Lost on failover: PostgreSQL 17 synchronizes only slots created with failover = true, and the subscription's parameters do not set it", + tone: 'degraded', + source: FAILOVER_SOURCE, + } +} + +export function cnpgLogicalPaths( + subscriptions: any[], + clusters: any[], + publications: any[], + poolers: any[], + /** Why Publications in a namespace are not readable (coverage), or null when they are. */ + publicationsUnavailable?: (namespace: string) => string | null, +): CNPGLogicalPath[] { + return subscriptions.map((sub) => { + const ns: string = sub?.metadata?.namespace ?? '' + const subscriber = clusters.find((c) => c?.metadata?.namespace === ns && c?.metadata?.name === sub?.spec?.cluster?.name) + const extName: string | undefined = sub?.spec?.externalClusterName + const ext = (subscriber?.spec?.externalClusters ?? []).find((e: any) => e?.name === extName) + const params = ext?.connectionParameters ?? {} + const publisher: CNPGPublisher = !subscriber + ? { kind: 'external', reason: 'the subscriber Cluster is not visible, so its external cluster cannot be read' } + : !ext + ? { kind: 'external', reason: `external cluster ${extName ?? '(unset)'} is not declared on ${subscriber.metadata?.name}` } + : cnpgResolvePublisher(params.host, ns, clusters, poolers) + const pubDb: string | undefined = sub?.spec?.publicationDBName || params.dbname + const pubName: string | undefined = sub?.spec?.publicationName + const pubObj = + publisher.kind === 'cluster' + ? publications.find( + (p) => + p?.metadata?.namespace === publisher.namespace && + p?.spec?.cluster?.name === publisher.name && + p?.spec?.name === pubName && + (!pubDb || p?.spec?.dbname === pubDb), + ) + : undefined + const slotParam: string | undefined = sub?.spec?.parameters?.slot_name + const createSlot = truthy(sub?.spec?.parameters?.create_slot) + const slot = + slotParam && slotParam.toLowerCase() === 'none' + ? { reason: 'slot_name = NONE: the subscription uses no slot' } + : slotParam + ? { name: slotParam } + : createSlot === false + ? { name: sub?.spec?.name, reason: 'create_slot = false: the slot must be created by hand' } + : { name: sub?.spec?.name } + return { + subscription: { + namespace: ns, + name: sub?.metadata?.name, + sqlName: sub?.spec?.name, + dbname: sub?.spec?.dbname, + cluster: sub?.spec?.cluster?.name, + applied: cnpgRoleState(sub) === 'pending' ? null : cnpgRoleState(sub) === 'applied', + message: sub?.status?.message || undefined, + }, + externalCluster: { name: extName, declared: !!ext, host: params.host, dbname: params.dbname }, + publisher, + publication: { + name: pubName, + dbname: pubDb, + object: pubObj + ? { namespace: pubObj.metadata.namespace, name: pubObj.metadata.name, applied: cnpgRoleState(pubObj) === 'pending' ? null : cnpgRoleState(pubObj) === 'applied' } + : undefined, + unavailable: !pubObj && publisher.kind === 'cluster' ? publicationsUnavailable?.(publisher.namespace) ?? undefined : undefined, + }, + slot, + failover: cnpgSlotFailover(publisher, sub), + } + }) +} + +/** What the publisher primary's instance manager says about one slot, in a shape independent of the runtime API. */ +export interface CNPGPublisherSlots { + /** partial: the report was capped or incomplete, so a slot missing from it may exist. */ + state: 'ok' | 'partial' | 'denied' | 'unavailable' | 'notRead' + reason?: string + /** The latest refresh failed; slots are from an earlier read. */ + stale?: boolean + slots?: { name: string; type?: string; active?: boolean; walStatus?: string; retainedBytes?: number; database?: string }[] +} + +const SLOT_SOURCE = "Publisher primary's instance manager (/pg/status replicationSlotsInfo) and exporter (retained WAL)" + +export function cnpgLogicalSlotFact(path: CNPGLogicalPath, observed: CNPGPublisherSlots): Fact { + if (!path.slot.name) return { text: path.slot.reason ?? 'No slot', tone: 'neutral' } + if (path.publisher.kind !== 'cluster') return { text: `Slot ${path.slot.name}: not observable (publisher outside this cluster's view)`, tone: 'unknown' } + if (observed.state === 'denied') return { text: `Slot ${path.slot.name}: no access (needs get pods/proxy on the publisher)`, tone: 'unknown', source: SLOT_SOURCE } + if ((observed.state !== 'ok' && observed.state !== 'partial') || !observed.slots) { + return { text: `Slot ${path.slot.name}: not read${observed.reason ? ` (${observed.reason})` : ''}`, tone: 'unknown', source: SLOT_SOURCE } + } + const s = observed.slots.find((x) => x.name === path.slot.name) + if (!s && (observed.state === 'partial' || observed.stale)) { + return { + text: `Slot ${path.slot.name}: not in the reported slots (${observed.stale ? 'from an earlier read; the latest refresh failed' : `report incomplete${observed.reason ? `: ${observed.reason}` : ''}`})`, + tone: 'unknown', + source: SLOT_SOURCE, + } + } + if (!s) { + return { + text: `Slot ${path.slot.name} not found on the publisher primary${path.slot.reason ? `: ${path.slot.reason}` : ''}`, + tone: path.subscription.applied === true ? 'degraded' : 'unknown', + source: SLOT_SOURCE, + } + } + const parts = [`Slot ${s.name}`, s.type ?? 'type unknown', s.active === undefined ? 'activity unknown' : s.active ? 'active' : 'inactive'] + if (s.retainedBytes !== undefined) parts.push(`retains ${cnpgFormatBytes(s.retainedBytes)} of WAL`) + if (s.walStatus) parts.push(`WAL ${s.walStatus}`) + const bad = s.active === false || s.walStatus === 'lost' || s.walStatus === 'unreserved' + if (observed.stale) { + return { text: `${parts.join(' · ')} (from an earlier read; the latest refresh failed)`, tone: 'unknown', source: SLOT_SOURCE } + } + return { text: parts.join(' · '), tone: s.walStatus === 'lost' ? 'unhealthy' : bad ? 'degraded' : 'healthy', source: SLOT_SOURCE } +} + +/** "cluster/database", with an unknown database said in words rather than as "?". */ +export function cnpgLogicalLocation(where: string, dbname: string | undefined): string { + return dbname ? `${where}/${dbname}` : `${where} · database unknown` +} diff --git a/packages/k8s-ui/src/components/cnpg/pooler.test.ts b/packages/k8s-ui/src/components/cnpg/pooler.test.ts new file mode 100644 index 0000000000..9ca8abf8c5 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/pooler.test.ts @@ -0,0 +1,84 @@ +import { describe, expect, it } from 'vitest' +import { poolerPressureCoverage, poolerPressureFact, aggregatePoolerPools, observedPause, poolerBackendService, poolerReadiness, poolerPodPressure } from './pooler' + +describe('poolerReadiness', () => { + it('reads readiness from the Deployment, never from the scheduled count', () => { + expect(poolerReadiness({ name: 'p', state: 'ok', replicas: 2, readyReplicas: 2 })).toMatchObject({ text: '2/2 ready', level: 'healthy' }) + expect(poolerReadiness({ name: 'p', state: 'ok', replicas: 2, readyReplicas: 1 }).level).toBe('degraded') + expect(poolerReadiness({ name: 'p', state: 'ok', replicas: 2, readyReplicas: 0 }).level).toBe('unhealthy') + }) + it('says unknown, not failed, when the Deployment is unreadable', () => { + expect(poolerReadiness({ name: 'p', state: 'unreadable' })).toMatchObject({ text: 'Unknown', level: 'unknown' }) + expect(poolerReadiness(undefined).level).toBe('unknown') + }) +}) + +describe('aggregatePoolerPools', () => { + it('sums each pool over the Pods that reported it and keeps unreported fields undefined', () => { + const rows = aggregatePoolerPools([ + { pod: 'a', state: 'ok', pools: [{ database: 'app', user: 'app', clActive: 3, clWaiting: 1, maxwaitSeconds: 0.5, poolMode: 'transaction' }] }, + { pod: 'b', state: 'ok', pools: [{ database: 'app', user: 'app', clActive: 2, clWaiting: 0, maxwaitSeconds: 2, poolMode: 'transaction' }] }, + ]) + expect(rows).toHaveLength(1) + expect(rows[0]).toMatchObject({ clActive: 5, clWaiting: 1, maxwaitSeconds: 2, pods: 2, poolModes: ['transaction'] }) + expect(rows[0].svActive).toBeUndefined() + }) +}) + +describe('observedPause', () => { + it('distinguishes all, some and unreadable', () => { + expect(observedPause({ state: 'ok', pods: [{ pod: 'a', state: 'ok', paused: true }, { pod: 'b', state: 'ok', paused: true }] })?.text).toBe('Paused on 2 of 2 PgBouncers') + expect(observedPause({ state: 'ok', pods: [{ pod: 'a', state: 'ok', paused: true }, { pod: 'b', state: 'ok', paused: false }] })?.level).toBe('alert') + expect(observedPause({ state: 'ok', pods: [{ pod: 'a', state: 'ok', paused: false }, { pod: 'b', state: 'error' }] })?.text).toBe('Serving (not paused) on 1 of 2 PgBouncers · 1 not read: b (error)') + expect(observedPause({ state: 'denied', grant: { verb: 'create', resource: 'pods', subresource: 'exec', namespace: 'x' }, pods: [] })?.text).toContain('create pods/exec') + }) +}) + +describe('poolerBackendService', () => { + it('names the Cluster Service by type', () => { + expect(poolerBackendService('pg', 'rw')).toBe('pg-rw') + expect(poolerBackendService('pg', 'ro')).toBe('pg-ro') + expect(poolerBackendService('pg', undefined)).toBeUndefined() + }) +}) + +describe('poolerPodPressure', () => { + it('totals each Pod on its own so one queuing Pod is visible', () => { + const rows = poolerPodPressure([ + { pod: 'pooler-b', state: 'ok', pools: [{ database: 'app', user: 'app', clActive: 2, clWaiting: 0, svActive: 1 }, { database: 'app', user: 'ro', clActive: 1, clWaiting: 0, svActive: 0 }] }, + { pod: 'pooler-a', state: 'ok', pools: [{ database: 'app', user: 'app', clActive: 20, clWaiting: 15, svActive: 20, maxwaitSeconds: 4.2 }] }, + { pod: 'pooler-c', state: 'unreachable', error: 'no answer within 5s' }, + ] as never) + expect(rows).toEqual([ + { pod: 'pooler-a', state: 'ok', error: undefined, clActive: 20, clWaiting: 15, svActive: 20, maxwaitSeconds: 4.2 }, + { pod: 'pooler-b', state: 'ok', error: undefined, clActive: 3, clWaiting: 0, svActive: 1 }, + { pod: 'pooler-c', state: 'unreachable', error: 'no answer within 5s' }, + ]) + }) +}) + +describe('pooler pressure certainty', () => { + it('says idle only when every Pod answered in full', () => { + const a = { pod: 'a', state: 'ok', pools: [] } + expect(poolerPressureCoverage([a]).empty).toBe('Idle: no client pools open') + for (const state of ['partial', 'unreachable']) { + const c = poolerPressureCoverage([a, { pod: 'b', state, reason: 'pool list capped' }]) + expect(c.complete).toBe(false) + expect(c.empty).toBe('No pools seen in what was read') + expect(c.limitation).toContain('b:') + expect(c.limitation).toContain('pool list capped') + } + }) + it('shows unknown for partial zero and a lower bound for partial positive counts', () => { + const p = { pod: 'a', state: 'partial', pools: [{ database: 'app', user: 'app', clWaiting: 0, clActive: 3 }] } + expect(poolerPressureFact([p], 'clWaiting')).toMatchObject({ text: 'Unknown', tone: 'unknown' }) + expect(poolerPressureFact([p], 'clActive')).toMatchObject({ text: '≥3', tone: 'unknown' }) + expect(poolerPressureFact([{ ...p, state: 'ok' }], 'clWaiting').text).toBe('0') + expect(poolerPressureFact([{ ...p, state: 'ok' }, { pod: 'b', state: 'unreachable' }], 'clWaiting').text).toBe('Unknown') + }) + it('qualifies a total when a pool omitted the field, even with complete Pod reads', () => { + const pods = [{ pod: 'a', state: 'ok', pools: [{ database: 'app', user: 'a', clActive: 2 }, { database: 'app', user: 'b' }] }] + expect(poolerPressureFact(pods, 'clActive').text).toBe('≥2') + expect(poolerPressureFact(pods, 'clWaiting').text).toBe('Unknown') + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/pooler.ts b/packages/k8s-ui/src/components/cnpg/pooler.ts new file mode 100644 index 0000000000..ee4041cbc4 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/pooler.ts @@ -0,0 +1,225 @@ +import type { Fact } from '../facts' +import type { HealthLevel } from '../resources/resource-utils' +import { formatGrant, type Grant } from '../../utils/grant' + +/** The Deployment a Pooler runs, as the host read it. */ +export interface CNPGPoolerDeploymentLive { + name: string + state: 'ok' | 'missing' | 'unreadable' | 'foreign' + replicas?: number + readyReplicas?: number + updatedReplicas?: number + availableReplicas?: number +} + +export interface CNPGPoolerServiceLive { + name: string + state: 'ok' | 'missing' | 'unreadable' | 'foreign' + type?: string + port?: number +} + +export interface CNPGPoolerPoolSample { + database: string + user: string + clActive?: number + clWaiting?: number + svActive?: number + svIdle?: number + svUsed?: number + maxwaitSeconds?: number + poolMode?: string +} + +/** Live PgBouncer exporter reads, one per pooler Pod. */ +export interface CNPGPoolerPressureLive { + state: 'loading' | 'denied' | 'error' | 'ok' + reason?: string + pods: { pod: string; state: string; error?: string; reason?: string; schedulingReason?: string; pools?: CNPGPoolerPoolSample[] }[] +} + +/** Each PgBouncer's own SHOW STATE. */ +export interface CNPGPoolerObservedLive { + state: 'loading' | 'denied' | 'error' | 'ok' + grant?: Grant + reason?: string + pods: { pod: string; state: string; paused?: boolean; error?: string }[] +} + +/** + * What a host that reads the cluster adds to the Pooler summary. Each part is + * optional: a part the host does not provide keeps the resource-only wording. + */ +export interface CNPGPoolerLive { + deployment?: CNPGPoolerDeploymentLive + service?: CNPGPoolerServiceLive + pressure?: CNPGPoolerPressureLive + observed?: CNPGPoolerObservedLive +} + +export interface CNPGPoolerReadiness { + text: string + level: HealthLevel + detail?: string +} + +/** Readiness from the Pooler's Deployment; the Pooler's status only counts scheduled Pods. */ +export function poolerReadiness(d: CNPGPoolerDeploymentLive | undefined): CNPGPoolerReadiness { + if (!d) return { text: 'Unknown', level: 'unknown', detail: 'Readiness is on the Pooler’s Deployment, which was not read' } + switch (d.state) { + case 'missing': + return { text: 'No Deployment', level: 'unhealthy', detail: `Deployment ${d.name} does not exist` } + case 'unreadable': + return { text: 'Unknown', level: 'unknown', detail: `No access to Deployment ${d.name}` } + case 'foreign': + return { text: 'Unknown', level: 'unknown', detail: `Deployment ${d.name} is not controlled by this Pooler` } + } + const want = d.replicas ?? 0 + const ready = d.readyReplicas ?? 0 + const detail = `from Deployment ${d.name}` + if (want === 0) return { text: 'Scaled to zero', level: 'neutral', detail } + if (ready === 0) return { text: `0/${want} ready`, level: 'unhealthy', detail } + if (ready < want) return { text: `${ready}/${want} ready`, level: 'degraded', detail } + return { text: `${ready}/${want} ready`, level: 'healthy', detail } +} + +export interface CNPGPoolerPoolRow { + database: string + user: string + poolModes: string[] + clActive?: number + clWaiting?: number + svActive?: number + svIdle?: number + svUsed?: number + maxwaitSeconds?: number + /** Pods whose exporter reported this pool. */ + pods: number +} + +/** + * One row per database/user pool, summed over the Pods that reported it + * (max wait is the maximum). A field no Pod reported stays undefined. + */ +export function aggregatePoolerPools(pods: CNPGPoolerPressureLive['pods']): CNPGPoolerPoolRow[] { + const rows = new Map() + const add = (a: number | undefined, b: number | undefined) => (b === undefined ? a : (a ?? 0) + b) + for (const p of pods) { + for (const pool of p.pools ?? []) { + const key = `${pool.database}\u0000${pool.user}` + const row = rows.get(key) ?? { database: pool.database, user: pool.user, poolModes: [], pods: 0 } + row.pods++ + row.clActive = add(row.clActive, pool.clActive) + row.clWaiting = add(row.clWaiting, pool.clWaiting) + row.svActive = add(row.svActive, pool.svActive) + row.svIdle = add(row.svIdle, pool.svIdle) + row.svUsed = add(row.svUsed, pool.svUsed) + if (pool.maxwaitSeconds !== undefined) row.maxwaitSeconds = Math.max(row.maxwaitSeconds ?? 0, pool.maxwaitSeconds) + if (pool.poolMode && !row.poolModes.includes(pool.poolMode)) row.poolModes.push(pool.poolMode) + rows.set(key, row) + } + } + return [...rows.values()].sort((a, b) => a.database.localeCompare(b.database) || a.user.localeCompare(b.user)) +} + +export interface CNPGPoolerPodRow { + pod: string + /** ok | partial | the read's failure state. */ + state: string + error?: string + clActive?: number + clWaiting?: number + svActive?: number + maxwaitSeconds?: number +} + +/** + * Each PgBouncer Pod's own totals across its pools, so one saturated Pod is + * not hidden in the sum. A Pod that did not report keeps its state and no + * numbers; a field none of its pools reported stays undefined. + */ +export function poolerPodPressure(pods: CNPGPoolerPressureLive['pods']): CNPGPoolerPodRow[] { + const add = (a: number | undefined, b: number | undefined) => (b === undefined ? a : (a ?? 0) + b) + return pods + .map((p) => { + const row: CNPGPoolerPodRow = { pod: p.pod, state: p.state, error: p.error } + for (const pool of p.pools ?? []) { + row.clActive = add(row.clActive, pool.clActive) + row.clWaiting = add(row.clWaiting, pool.clWaiting) + row.svActive = add(row.svActive, pool.svActive) + if (pool.maxwaitSeconds !== undefined) row.maxwaitSeconds = Math.max(row.maxwaitSeconds ?? 0, pool.maxwaitSeconds) + } + return row + }) + .sort((a, b) => a.pod.localeCompare(b.pod)) +} + +/** + * PgBouncer settings worth showing, with the value PgBouncer uses when the + * Pooler leaves one unset. Defaults are named only where they were read from + * PgBouncer itself (SHOW CONFIG on PgBouncer 1.24); CloudNativePG writes only + * the parameters the Pooler sets. + */ +export const POOLER_LIMIT_PARAMETERS: { key: string; label: string; pgbouncerDefault?: string }[] = [ + { key: 'default_pool_size', label: 'Pool size (per database/user)', pgbouncerDefault: '20' }, + { key: 'max_client_conn', label: 'Max client connections', pgbouncerDefault: '100' }, + { key: 'max_db_connections', label: 'Max connections per database', pgbouncerDefault: '0 (unlimited)' }, + { key: 'max_user_connections', label: 'Max connections per user', pgbouncerDefault: '0 (unlimited)' }, + { key: 'reserve_pool_size', label: 'Reserve pool', pgbouncerDefault: '0' }, + { key: 'min_pool_size', label: 'Min pool size', pgbouncerDefault: '0' }, +] + +/** The Cluster Service PgBouncer forwards to, by the Pooler's type. */ +export function poolerBackendService(cluster: string | undefined, type: string | undefined): string | undefined { + if (!cluster || !type) return undefined + if (type === 'rw' || type === 'ro' || type === 'r') return `${cluster}-${type}` + return undefined +} + +/** Observed pause across PgBouncers: every one, none, some, or unknown. */ +export function observedPause(o: CNPGPoolerObservedLive | undefined): { text: string; level: HealthLevel } | null { + if (!o) return null + if (o.state === 'loading') return { text: 'Reading…', level: 'unknown' } + if (o.state === 'denied') return { text: `Not observable: needs ${formatGrant(o.grant) ?? 'create pods/exec'}`, level: 'unknown' } + if (o.state === 'error') return { text: `Not observable: ${o.reason ?? 'read failed'}`, level: 'unknown' } + if (o.pods.length === 0) return { text: 'No PgBouncer Pods', level: 'unknown' } + const read = o.pods.filter((p) => p.state === 'ok' && p.paused !== undefined) + const paused = read.filter((p) => p.paused).length + const unread = o.pods.length - read.length + const tail = unread > 0 ? ` · ${unread} not read: ${o.pods.filter((p) => p.state !== 'ok' || p.paused === undefined).map((p) => `${p.pod} (${p.error ?? p.state})`).join('; ')}` : '' + if (read.length === 0) return { text: `Not observable${tail}`, level: 'unknown' } + if (paused === 0) return { text: `Serving (not paused) on ${read.length} of ${o.pods.length} PgBouncers${tail}`, level: unread ? 'unknown' : 'healthy' } + if (paused === read.length) return { text: `Paused on ${paused} of ${o.pods.length} PgBouncers${tail}`, level: 'degraded' } + return { text: `Paused on ${paused} of ${o.pods.length} PgBouncers${tail}`, level: 'alert' } +} + +export function poolerPressureCoverage(pods: CNPGPoolerPressureLive['pods']) { + const reporting = pods.filter((p) => p.state === 'ok' || p.state === 'partial') + const complete = pods.length > 0 && pods.every((p) => p.state === 'ok') + const gaps = pods.filter((p) => p.state !== 'ok').map((p) => { + const reason = [p.reason, p.error].filter(Boolean).join(' · ') + return `${p.pod}: ${p.state === 'partial' ? 'partial' : 'not read'}${reason ? ` (${reason})` : p.state === 'partial' ? '' : ` (${p.state})`}` + }) + return { + reporting, + complete, + limitation: gaps.join('; ') || (pods.length === 0 ? 'No PgBouncer Pods answered' : undefined), + empty: complete ? 'Idle: no client pools open' : 'No pools seen in what was read', + } +} + +type PoolerMetric = 'clActive' | 'clWaiting' | 'svActive' | 'svIdle' | 'svUsed' | 'maxwaitSeconds' + +export function poolerPressureFact(pods: CNPGPoolerPressureLive['pods'], field: PoolerMetric, pool?: { database: string; user: string }): Fact { + const coverage = poolerPressureCoverage(pods) + const pools = coverage.reporting.flatMap((p) => p.pools ?? []).filter((p) => !pool || (p.database === pool.database && p.user === pool.user)) + const values = pools.map((p) => p[field]).filter((v): v is number => v !== undefined) + const complete = coverage.complete && values.length === pools.length + const value = field === 'maxwaitSeconds' ? Math.max(0, ...values) : values.reduce((a, b) => a + b, 0) + const label = { clActive: 'Active clients', clWaiting: 'Waiting clients', svActive: 'Active servers', svIdle: 'Idle servers', svUsed: 'Used servers', maxwaitSeconds: 'Max wait' }[field] + if ((!complete && value === 0) || (pools.length > 0 && values.length === 0)) { + return { text: 'Unknown', tone: 'unknown', source: coverage.limitation ?? `${label} not reported for every pool` } + } + const text = field === 'maxwaitSeconds' ? `${value.toFixed(1)} s` : String(value) + return { text: `${complete ? '' : '≥'}${text}`, tone: field === 'clWaiting' && value > 0 ? 'degraded' : complete ? 'neutral' : 'unknown', source: complete ? undefined : coverage.limitation ?? `${label} not reported for every pool` } +} diff --git a/packages/k8s-ui/src/components/cnpg/primitives.tsx b/packages/k8s-ui/src/components/cnpg/primitives.tsx index 7020a4da81..2c753cb1f6 100644 --- a/packages/k8s-ui/src/components/cnpg/primitives.tsx +++ b/packages/k8s-ui/src/components/cnpg/primitives.tsx @@ -1,134 +1,12 @@ -import type { ReactNode } from 'react' import { clsx } from 'clsx' -import type { HealthLevel } from '../resources/resource-utils' -import { formatAge } from '../resources/resource-utils' -import { StatusDot } from '../ui/status-tone' -import { Tooltip } from '../ui/Tooltip' -import { AlertBanner } from '../ui/drawer-components' -import { TONE_TEXT_CLASS } from '../ui/severity-tone' -import type { CNPGFact, CNPGProblem } from './workspace' +import { toneTextClass } from '../ui/status-tone' -export interface CNPGRef { - kind: string - group?: string - namespace: string - name: string -} - -export type CNPGNavigate = (ref: CNPGRef) => void - -const TONE_TEXT: Record = { - healthy: 'text-theme-text-primary', - degraded: TONE_TEXT_CLASS.amber, - alert: TONE_TEXT_CLASS.orange, - unhealthy: TONE_TEXT_CLASS.red, - unknown: 'text-theme-text-tertiary', - neutral: 'text-theme-text-secondary', -} - -export const CNPG_PRIMARY_BUTTON = 'btn-brand inline-flex items-center gap-1.5 px-3 py-1.5 text-sm font-medium' -export const CNPG_SECONDARY_BUTTON = - 'inline-flex items-center gap-1.5 rounded-lg border border-theme-border bg-theme-surface px-3 py-1.5 text-sm text-theme-text-primary transition-colors hover:bg-theme-hover' - -export function toneTextClass(tone: HealthLevel): string { - return TONE_TEXT[tone] -} - -export function FactValue({ fact, className }: { fact: CNPGFact; className?: string }) { - const age = fact.at ? formatAge(fact.at) : null - const body = ( - - {fact.text} - {age && {fact.text ? ' · ' : ''}{age} ago} - - ) - if (!fact.source && !fact.at) return body - return ( - - {body} - - ) -} - -export function FactSource({ fact }: { fact: CNPGFact }) { - if (!fact.source) return null - return
{fact.source}
-} - -export function FactGrid({ children }: { children: ReactNode }) { - return
{children}
-} - -export function FactRow({ label, children }: { label: ReactNode; children: ReactNode }) { +/** status.currentPrimary and the primary role label disagree: both are named rather than one silently winning. */ +export function PrimaryConflictNote({ conflict }: { conflict: { status: string; labelled: string } }) { return ( - <> -
{label}
-
{children}
- - ) -} - -export function SummaryHeading({ children, hint }: { children: ReactNode; hint?: ReactNode }) { - return ( -
-

{children}

- {hint && {hint}} +
+ CNPG status says primary {conflict.status}; the Pod labelled primary is{' '} + {conflict.labelled}. Status may be stale, or a failover is under way.
) } - -export function RefLink({ refTo, onNavigate, children, mono }: { refTo: CNPGRef; onNavigate?: CNPGNavigate; children?: ReactNode; mono?: boolean }) { - const label = children ?? refTo.name - if (!onNavigate) return {label} - return ( - - ) -} - -const PROBLEM_VARIANT: Record = { - critical: 'error', - warning: 'warning', - posture: 'info', -} - -export function ProblemCallout({ - problem, - more, - onNavigate, - action, - subjectIsSelf, -}: { - problem: CNPGProblem - more?: ReactNode - onNavigate?: CNPGNavigate - action?: ReactNode - /** The callout sits on the subject's own page, so linking to it would loop. */ - subjectIsSelf?: boolean -}) { - const aboutChild = !subjectIsSelf && problem.subject.kind !== 'Cluster' - return ( - -
- {aboutChild && ( - - {problem.subject.kind}{' '} - - - )} - {problem.source === 'audit' ? 'Radar check' : 'Radar issue'} - {action} - {more} -
-
- ) -} - -export function ToneDot({ tone }: { tone: HealthLevel }) { - return -} diff --git a/packages/k8s-ui/src/components/cnpg/relations.test.ts b/packages/k8s-ui/src/components/cnpg/relations.test.ts index 437dafbc70..c98b3d5f9f 100644 --- a/packages/k8s-ui/src/components/cnpg/relations.test.ts +++ b/packages/k8s-ui/src/components/cnpg/relations.test.ts @@ -1,11 +1,12 @@ import { describe, expect, it } from 'vitest' import { + cnpgScheduleDestinationBlocker, + cnpgArchiveMatchesCluster, appliedFact, backupDestination, backupsForScheduledBackup, clustersUsingCatalog, databaseForDeclaration, - gitopsSourceOf, inferredObjectStoreHealth, isBackupFromSchedule, issuesForObject, @@ -16,6 +17,23 @@ import { scheduledBackupOf, usersOfObjectStore, } from './relations' + +it('recognizes archive origin aliases without merging distinct paths', () => { + const cluster = { metadata: { name: 'pg' }, spec: { backup: { barmanObjectStore: { destinationPath: 's3://bucket/path', endpointURL: 'https://storage.example' } } } } + const source = { kind: 'inTree' as const, serverName: 'pg', barmanObjectStore: { destinationPath: 's3://BUCKET/path/', endpointURL: 'https://STORAGE.EXAMPLE:443/' } } + expect(cnpgArchiveMatchesCluster(cluster, source)).toBe(true) + expect(cnpgArchiveMatchesCluster(cluster, { ...source, barmanObjectStore: { ...source.barmanObjectStore, destinationPath: 's3://bucket/other' } })).toBe(false) + expect(cnpgArchiveMatchesCluster(cluster, { ...source, barmanObjectStore: { ...source.barmanObjectStore, endpointURL: 'https://storage.example/prefix' } })).toBe(false) +}) + +it('checks the destination for the schedule method and keeps unread targets unknown', () => { + const cluster = { apiVersion: 'postgresql.cnpg.io/v1', kind: 'Cluster', metadata: { name: 'payments', namespace: 'pg' }, spec: { backup: { volumeSnapshot: {} }, plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } } + const schedule = { metadata: { namespace: 'pg' }, spec: { cluster: { name: 'payments' } } } + expect(cnpgScheduleDestinationBlocker(schedule, [cluster])).toBe('No barmanObjectStore destination') + expect(cnpgScheduleDestinationBlocker(schedule, [])).toBeNull() + expect(cnpgScheduleDestinationBlocker({ ...schedule, spec: { ...schedule.spec, method: 'volumeSnapshot' } }, [cluster])).toBeNull() + expect(cnpgScheduleDestinationBlocker({ ...schedule, spec: { ...schedule.spec, method: 'plugin', pluginConfiguration: { name: 'barman-cloud.cloudnative-pg.io' } } }, [cluster])).toBeNull() +}) import { CNPG_WORKSPACE_KEYS, type CNPGWorkspaceIssue, type CNPGWorkspaceResponse } from './workspace' const PG = 'postgresql.cnpg.io/v1' @@ -104,9 +122,10 @@ describe('scheduledBackupOf / backupsForScheduledBackup', () => { describe('objectStoreForBackup / backupDestination', () => { const clusters = [pluginCluster('main', 'store-a')] - it('prefers the store the backup recorded', () => { + it('ignores the Backup parameter and infers the store from the current Cluster', () => { const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN, parameters: { barmanObjectName: 'store-b' } } } }) - expect(objectStoreForBackup(b, clusters)).toEqual({ name: 'store-b', inferred: false }) + expect(objectStoreForBackup(b, clusters)).toEqual({ name: 'store-a', inferred: true }) + expect(backupDestination(b, clusters)).toEqual({ type: 'objectStore', name: 'store-a', inferred: true }) }) it("marks a store taken from the Cluster's current plugin as inferred", () => { @@ -115,6 +134,14 @@ describe('objectStoreForBackup / backupDestination', () => { expect(backupDestination(b, clusters)).toEqual({ type: 'objectStore', name: 'store-a', inferred: true }) }) + it('cannot resolve a Backup parameter without a configured, readable target Cluster', () => { + const b = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN, parameters: { barmanObjectName: 'store-b' } } } }) + for (const targets of [[], [cluster('main')], [pluginCluster('other', 'store-a')], [{ ...pluginCluster('main', 'store-a'), metadata: { name: 'main', namespace: 'other' } }], [cluster('main', 'pg', { plugins: [{ name: PLUGIN, enabled: false, parameters: { barmanObjectName: 'store-a' } }] })]]) { + expect(objectStoreForBackup(b, targets)).toBeNull() + expect(backupDestination(b, targets)).toEqual({ type: 'unknown' }) + } + }) + it('checks plugin identity before reading barmanObjectName', () => { const other = backup('b', { spec: { method: 'plugin', pluginConfiguration: { name: 'other.example.com', parameters: { barmanObjectName: 'store-b' } } } }) expect(objectStoreForBackup(other, clusters)).toBeNull() @@ -129,6 +156,23 @@ describe('objectStoreForBackup / backupDestination', () => { }) }) +describe('barman-cloud schedule destinations', () => { + const schedule = { apiVersion: PG, kind: 'ScheduledBackup', metadata: { name: 'nightly', namespace: 'pg' }, spec: { cluster: { name: 'main' }, method: 'plugin', pluginConfiguration: { name: PLUGIN, parameters: { barmanObjectName: 'schedule-store', serverName: 'schedule-server' } } } } + + it('blocks a schedule override when the Cluster plugin has no destination', () => { + const target = cluster('main', 'pg', { plugins: [{ name: PLUGIN }] }) + expect(cnpgScheduleDestinationBlocker(schedule, [target])).toBe('No backup destination') + expect(objectStoreForBackup(schedule, [target])).toBeNull() + }) + + it('uses the configured Cluster destination despite a conflicting schedule override', () => { + const target = pluginCluster('main', 'cluster-store') + expect(cnpgScheduleDestinationBlocker(schedule, [target])).toBeNull() + expect(objectStoreForBackup(schedule, [target])).toEqual({ name: 'cluster-store', inferred: true }) + expect(cnpgScheduleDestinationBlocker(schedule, [cluster('main', 'pg', { plugins: [{ name: PLUGIN, enabled: false, parameters: { barmanObjectName: 'cluster-store' } }] })])).toBe('No backup destination') + }) +}) + describe('ObjectStore users and inferred health', () => { const store = { apiVersion: BARMAN, @@ -254,16 +298,6 @@ describe('declarations', () => { expect(missingManagedRole({ status: { message: 'connection refused' } }, cluster('main'))).toBeNull() }) - it('reads the GitOps owner labels', () => { - expect(gitopsSourceOf({ metadata: { labels: { 'argocd.argoproj.io/instance': 'app' } } })).toEqual({ tool: 'argocd', name: 'app' }) - expect(gitopsSourceOf({ metadata: { labels: { 'kustomize.toolkit.fluxcd.io/name': 'k', 'kustomize.toolkit.fluxcd.io/namespace': 'flux' } } })).toEqual({ - tool: 'flux', - name: 'k', - namespace: 'flux', - }) - expect(gitopsSourceOf({ metadata: {} })).toBeNull() - }) - const decl = (kind: string, name: string, clusterName: string, dbname: string, ns = 'pg') => ({ apiVersion: PG, kind, @@ -302,3 +336,19 @@ describe('relationUnavailable', () => { expect(relationUnavailable(null, 'backups', 'pg', 'Backups')).toBe('Backups could not be read') }) }) + +it('does not apply declaration results from an earlier spec', () => { + for (const applied of [true, false]) { + expect(appliedFact({ metadata: { generation: 3 }, status: { applied, observedGeneration: 2 } })).toEqual({ text: 'Pending · awaiting the operator for the current spec', tone: 'unknown' }) + } + expect(appliedFact({ metadata: { generation: 3 }, status: { applied: true, observedGeneration: 3 } }).tone).toBe('healthy') +}) + + +it('never infers a plugin archive from a replacement Cluster', () => { + const target = { ...pluginCluster('main', 'new-store'), metadata: { name: 'main', namespace: 'pg', uid: 'new', creationTimestamp: '2026-10-01T00:00:00Z' } } + for (const status of [{ pluginMetadata: { clusterUID: 'old' } }, { startedAt: '2026-09-30T00:00:00Z' }]) { + const old = backup('old', { spec: { method: 'plugin', pluginConfiguration: { name: PLUGIN } }, status }) + expect(objectStoreForBackup(old, [target])).toBeNull() + } +}) diff --git a/packages/k8s-ui/src/components/cnpg/relations.ts b/packages/k8s-ui/src/components/cnpg/relations.ts index 2db316d662..179661f9b1 100644 --- a/packages/k8s-ui/src/components/cnpg/relations.ts +++ b/packages/k8s-ui/src/components/cnpg/relations.ts @@ -1,9 +1,9 @@ +import { normalizeURLForComparison } from '../../utils/url-path' +import { cnpgBackupDeclaration, cnpgBackupDestinationBlocker, cnpgBackupBlockerText } from '../../utils/cnpg-backup' // Pure relationship lookups between CloudNativePG objects in the workspace // payload. Each helper answers only from what the objects record; a relation // that cannot be established returns null or an empty list, never a guess. -import type { BadgeSeverity } from '../ui/Badge' -import type { HealthLevel } from '../resources/resource-utils' import { CNPG_BARMAN_PLUGIN_NAME, CNPG_GROUP, @@ -12,15 +12,8 @@ import { isApiGroup, type CNPGObjectStoreRecoveryWindow, } from '../resources/resource-utils-cnpg' -import { - cnpgIssueCategory, - coverageReadable, - type CNPGFact, - type CNPGProblem, - type CNPGWorkspaceIssue, - type CNPGWorkspaceKey, - type CNPGWorkspaceResponse, -} from './workspace' +import { cnpgIssueCategory, cnpgIssueOrigin, cnpgIssueText, cnpgCoverageGap, coverageReadable, type CNPGProblem, type CNPGWorkspaceIssue, type CNPGWorkspaceKey, type CNPGWorkspaceResponse } from './workspace' +import { type Fact } from '../facts' export interface CNPGObjectRef { kind: string @@ -58,21 +51,6 @@ export function refOf(obj: any, kind: string, group: string = CNPG_GROUP): CNPGO return { kind, group, namespace: nsOf(obj), name: nameOf(obj) } } -export function healthSeverity(level: HealthLevel): BadgeSeverity { - switch (level) { - case 'healthy': - return 'success' - case 'unhealthy': - return 'error' - case 'alert': - return 'alert' - case 'degraded': - return 'warning' - default: - return 'neutral' - } -} - // --------------------------------------------------------------------------- // Workspace access // --------------------------------------------------------------------------- @@ -94,17 +72,7 @@ export function relationUnavailable( if (!ws) return `${what} could not be read` const cov = ws.coverage?.[key] ?? { state: 'notInstalled' as const } if (coverageReadable(cov, namespace)) return null - switch (cov.state) { - case 'denied': - case 'partial': - return `No access to ${what}` - case 'syncing': - return 'Loading…' - case 'error': - return `Could not read ${what}` - default: - return `${what} are not installed` - } + return cnpgCoverageGap(cov, what, namespace, `${what} are not installed`) } export function clustersIn(ws: CNPGWorkspaceResponse | null | undefined): any[] { @@ -115,7 +83,8 @@ export function clustersIn(ws: CNPGWorkspaceResponse | null | undefined): any[] export function targetCluster(obj: any, clusters: any[]): any | null { const name = specCluster(obj) if (!name) return null - return clusters.find((c) => isCNPGKind(c, 'Cluster') && nsOf(c) === nsOf(obj) && nameOf(c) === name) ?? null + const cluster = clusters.find((c) => isCNPGKind(c, 'Cluster') && nsOf(c) === nsOf(obj) && nameOf(c) === name) ?? null + return isCNPGKind(obj, 'Backup') && !cnpgBackupMatchesCluster(obj, cluster) ? null : cluster } // --------------------------------------------------------------------------- @@ -140,10 +109,10 @@ export function problemsForObject(issues: CNPGWorkspaceIssue[] | undefined, ref: id: `${issue.id}:${issue.kind}/${issue.name}`, severity: issue.severity, category: cnpgIssueCategory(issue), - title: issue.message || issue.reason, - detail: issue.cause || undefined, + ...cnpgIssueText(issue), subject: { kind: issue.kind, group: issue.group ?? '', namespace: issue.namespace ?? '', name: issue.name }, source: 'issue', + origin: cnpgIssueOrigin(issue), })) } @@ -195,20 +164,50 @@ export function backupTime(backup: any): number { return parseTime(backup?.status?.startedAt) || parseTime(backup?.metadata?.creationTimestamp) } +export function cnpgBackupMatchesCluster(backup: any, cluster: any): boolean { + if (!cluster || backup?.spec?.cluster?.name !== cluster.metadata?.name || backup.metadata?.namespace !== cluster.metadata?.namespace) return false + const owner = backup.metadata?.ownerReferences?.find((ref: any) => ref.kind === 'Cluster' && isApiGroup(ref.apiVersion, CNPG_GROUP)) + const recordedUID = backup.status?.pluginMetadata?.clusterUID || owner?.uid + if (recordedUID && recordedUID !== cluster.metadata?.uid) return false + const began = Date.parse(backup.status?.startedAt ?? backup.metadata?.creationTimestamp ?? '') + const created = Date.parse(cluster.metadata?.creationTimestamp ?? '') + return !(Number.isFinite(began) && Number.isFinite(created) && began < created) +} + +export type CNPGArchiveSource = + | { kind: 'objectStore'; objectStore: string; serverName: string } + | { kind: 'inTree'; barmanObjectStore: Record; serverName: string } + +export function cnpgArchiveMatchesCluster(cluster: any, source: CNPGArchiveSource): boolean { + if (source.kind === 'objectStore') { + const plugin = getCNPGClusterBarmanPlugin(cluster) + return plugin?.barmanObjectName === source.objectStore && + (plugin?.serverName || cluster.metadata?.name) === source.serverName + } + const archive = cluster.spec?.backup?.barmanObjectStore + const path = source.barmanObjectStore.destinationPath + const endpoint = source.barmanObjectStore.endpointURL + const destination = typeof path === 'string' ? normalizeURLForComparison(path) : null + const endpointKey = normalizeURLForComparison(typeof endpoint === 'string' ? endpoint : '') + return !!archive?.destinationPath && typeof path === 'string' && !!path && + destination !== null && endpointKey !== null && + (archive.serverName || cluster.metadata?.name) === source.serverName && + normalizeURLForComparison(archive.destinationPath) === destination && + normalizeURLForComparison(archive.endpointURL || '') === endpointKey +} + /** - * The ObjectStore a barman-cloud plugin Backup wrote to. The Backup's own - * plugin parameters are a record of that run; the target Cluster's plugin - * configuration is only what it is configured with now, so a store taken from - * there is marked inferred. + * Barman-cloud ignores Backup and ScheduledBackup parameters. The current + * Cluster plugin chooses the ObjectStore; it may have changed since a Backup + * ran, so the historical destination remains inferred. */ export function objectStoreForBackup(backup: any, clusters: any[]): { name: string; inferred: boolean } | null { const method = backup?.status?.method || backup?.spec?.method if (method !== 'plugin') return null const cfg = backup?.spec?.pluginConfiguration if (cfg?.name !== CNPG_BARMAN_PLUGIN_NAME) return null - const own = cfg?.parameters?.barmanObjectName - if (typeof own === 'string' && own) return { name: own, inferred: false } const cluster = targetCluster(backup, clusters) + if (backup.kind === 'Backup' && !cnpgBackupMatchesCluster(backup, cluster)) return null const current = cluster ? getCNPGClusterBarmanPlugin(cluster)?.barmanObjectName : undefined return current ? { name: current, inferred: true } : null } @@ -229,6 +228,14 @@ export function backupDestination(backup: any, clusters: any[]): CNPGBackupDesti return { type: 'unknown' } } +/** Whether the Cluster declares a destination for this schedule's method; unread Clusters stay unknown. */ +export function cnpgScheduleDestinationBlocker(schedule: any, clusters: any[]): string | null { + const cluster = targetCluster(schedule, clusters) + if (!cluster) return null + const blocker = cnpgBackupDestinationBlocker(cnpgBackupDeclaration(cluster), schedule.spec?.method || 'barmanObjectStore', schedule.spec?.pluginConfiguration?.name) + return blocker ? cnpgBackupBlockerText(blocker) : null +} + // --------------------------------------------------------------------------- // ObjectStore // --------------------------------------------------------------------------- @@ -256,16 +263,16 @@ export function usersOfObjectStore(store: any, clusters: any[]): CNPGObjectStore export interface CNPGObjectStoreEvidence { cluster: CNPGObjectRef serverName: string - archiving: CNPGFact + archiving: Fact window: CNPGObjectStoreRecoveryWindow | null } export interface CNPGObjectStoreHealth { - summary: CNPGFact + summary: Fact evidence: CNPGObjectStoreEvidence[] } -function archivingFact(cluster: any): CNPGFact { +function archivingFact(cluster: any): Fact { const conds = cluster?.status?.conditions const c = Array.isArray(conds) ? conds.find((x: any) => x?.type === 'ContinuousArchiving') : null if (!c) return { text: 'WAL archiving not reported', tone: 'unknown' } @@ -319,14 +326,17 @@ export function inferredObjectStoreHealth(store: any, users: CNPGObjectStoreUser // Declarative objects // --------------------------------------------------------------------------- -export function appliedFact(obj: any): CNPGFact { +export function appliedFact(obj: any): Fact { + if (observedGenerationFact(obj).tone === 'degraded') { + return { text: 'Pending · awaiting the operator for the current spec', tone: 'unknown' } + } const applied = obj?.status?.applied if (applied === true) return { text: 'Applied', tone: 'healthy' } if (applied === false) return { text: 'Not applied', tone: 'unhealthy' } return { text: 'Pending · the operator has not reported a result yet', tone: 'unknown' } } -export function observedGenerationFact(obj: any): CNPGFact { +export function observedGenerationFact(obj: any): Fact { const observed = obj?.status?.observedGeneration const generation = obj?.metadata?.generation if (typeof observed !== 'number') return { text: 'Not reported', tone: 'unknown' } @@ -351,7 +361,6 @@ export function missingManagedRole(obj: any, cluster: any | null): string | null return names.includes(m[1]) ? null : m[1] } -export { cnpgGitOpsSource as gitopsSourceOf } from './workspace' /** Publications and Subscriptions on the same Cluster and PostgreSQL database. */ export function replicationForDatabase( diff --git a/packages/k8s-ui/src/components/cnpg/schedule.ts b/packages/k8s-ui/src/components/cnpg/schedule.ts new file mode 100644 index 0000000000..0d93113cf5 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/schedule.ts @@ -0,0 +1,35 @@ +/** + * The server's reading of a ScheduledBackup schedule (see + * /api/cnpg/scheduledbackups/{ns}/{name}/schedule-preview): parsed as + * CloudNativePG parses it, with the next runs counted as the operator counts + * them. Times are RFC 3339 UTC. + */ +export interface CNPGSchedulePreview { + schedule: string + valid: boolean + error?: string + description?: string + nextRuns?: string[] + /** The first run is due already: the operator creates a backup as soon as it reconciles. */ + runsImmediately?: boolean + basis: 'lastCheckTime' | 'now' + lastCheckTime?: string + suspended?: boolean + clock?: { zone: string; declared: boolean; source: string } +} + +export function formatCNPGRunTime(iso: string): { utc: string; local: string } { + const d = new Date(iso) + const pad = (n: number) => String(n).padStart(2, '0') + const utc = `${d.getUTCFullYear()}-${pad(d.getUTCMonth() + 1)}-${pad(d.getUTCDate())} ${pad(d.getUTCHours())}:${pad(d.getUTCMinutes())}:${pad(d.getUTCSeconds())} UTC` + return { utc, local: d.toLocaleString() } +} + +/** One line for the runs list: why the first run is when it is. */ +export function cnpgScheduleBasisNote(p: CNPGSchedulePreview): string { + const clock = p.clock?.source ?? 'Operator clock is not established; upcoming times assume UTC.' + if (p.suspended) return `Suspended: nothing runs until it is resumed. ${clock}` + const basis = p.basis === 'lastCheckTime' ? "Counted from the operator's last check (status.lastCheckTime)." : 'Counted from now; the operator starts counting at its first check.' + const due = p.runsImmediately ? ` ${p.clock?.declared ? 'A run is due on the declared clock.' : 'A run is due in this UTC estimate.'} The operator takes at most one catch-up backup.` : '' + return `${basis} ${clock}${due}` +} diff --git a/packages/k8s-ui/src/components/cnpg/workspace-disk.test.ts b/packages/k8s-ui/src/components/cnpg/workspace-disk.test.ts new file mode 100644 index 0000000000..54850a0087 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/workspace-disk.test.ts @@ -0,0 +1,97 @@ +import { describe, it, expect } from 'vitest' +import { applyCNPGDisk, buildCNPGFleet, cnpgDiskFact, CNPG_WORKSPACE_KEYS, type CNPGDiskReading, type CNPGWorkspaceResponse } from './workspace' + +function cluster(name: string): any { + return { + apiVersion: 'postgresql.cnpg.io/v1', + kind: 'Cluster', + metadata: { name, namespace: 'db' }, + spec: { instances: 1 }, + status: { phase: 'Cluster in healthy state', readyInstances: 1, currentPrimary: `${name}-1` }, + } +} + +function fleetOf(...names: string[]) { + const coverage: CNPGWorkspaceResponse['coverage'] = {} + for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } + return buildCNPGFleet({ + installed: true, + context: 'test', + namespaces: null, + coverage, + objects: { clusters: names.map(cluster) }, + issues: [], + audit: [], + backupsOmitted: 0, + }) +} + +function reading(name: string, over: Partial = {}): CNPGDiskReading { + return { namespace: 'db', name, state: 'ok', claims: 1, measured: 1, ...over } +} + +const max = (ratio: number) => ({ claim: 'pg-1', instance: 'pg-1', role: 'PG_DATA', usedBytes: ratio * 10 * 1024 ** 3, capacityBytes: 10 * 1024 ** 3, ratio }) + +describe('cnpgDiskFact', () => { + it('never reads an unmeasured cluster as zero or healthy', () => { + for (const state of ['noSeries', 'noPrometheus', 'denied', 'error', 'notRead']) { + const f = cnpgDiskFact(reading('pg', { state, measured: 0 })) + expect(f.tone).toBe('unknown') + expect(f.text).not.toMatch(/\d/) + } + expect(cnpgDiskFact(undefined).tone).toBe('unknown') + }) + + it('names the fullest volume and its source, and labels partial coverage', () => { + const f = cnpgDiskFact(reading('pg', { state: 'partial', claims: 2, measured: 1, max: max(0.5) })) + expect(f.text).toBe('50% used') + expect(f.tone).toBe('healthy') + expect(f.source).toContain('data volume of pg-1') + expect(f.source).toContain('kubelet volume stats') + expect(f.source).toContain('1 of 2 volumes measured') + }) + + it('names the missing grant when denied', () => { + expect(cnpgDiskFact(reading('pg', { state: 'denied', grant: { verb: 'list', resource: 'persistentvolumeclaims', namespace: 'db' }, measured: 0 })).source).toBe('Needs list persistentvolumeclaims in namespace db') + }) +}) + +describe('applyCNPGDisk', () => { + it('says when the volume stats were matched to the cluster by claim name only', () => { + const note = "Matched by namespace and claim names. Radar couldn't confirm these volume stats belong to this exact cluster (no cluster label it could check)" + const fleet = applyCNPGDisk(fleetOf('pg-a'), [reading('pg-a', { max: max(0.95), isolation: { mode: 'unverified', note } })]) + const p = fleet.rows[0].problems[0] + expect(p).toMatchObject({ measuredBy: 'kubelet, matched by claim name', unverifiedMatch: true }) + expect(p.detail).toContain(note) + expect(fleet.rows[0].disk?.source).toContain(note) + }) + + it('puts low-disk clusters into Needs attention by severity', () => { + const fleet = applyCNPGDisk(fleetOf('pg-a', 'pg-b', 'pg-c'), [ + reading('pg-a', { max: max(0.85) }), + reading('pg-b', { max: max(0.95) }), + reading('pg-c', { max: max(0.4) }), + ]) + expect(fleet.attentionCount).toBe(2) + const a = fleet.rows.find((r) => r.name === 'pg-a')! + const b = fleet.rows.find((r) => r.name === 'pg-b')! + const c = fleet.rows.find((r) => r.name === 'pg-c')! + expect(a.problems[0].severity).toBe('warning') + expect(b.problems[0].severity).toBe('critical') + expect(b.problems[0].source).toBe('measurement') + expect(b.problems[0].title).toBe('The data volume of pg-1 is 95% full') + expect(c.attention).toBe(false) + expect(fleet.categoryCounts.availability).toBe(2) + }) + + it('raises nothing without a measurement', () => { + const fleet = applyCNPGDisk(fleetOf('pg-a'), [reading('pg-a', { state: 'noPrometheus', measured: 0, reason: 'Radar is not connected to Prometheus: x' })]) + expect(fleet.attentionCount).toBe(0) + expect(fleet.rows[0].disk).toMatchObject({ text: 'No usage metrics', source: 'Prometheus not connected', detail: 'Radar is not connected to Prometheus: x' }) + }) + + it('leaves the fleet as built when no reading was requested', () => { + const base = fleetOf('pg-a') + expect(applyCNPGDisk(base, undefined)).toBe(base) + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/workspace-fleet-metrics.test.ts b/packages/k8s-ui/src/components/cnpg/workspace-fleet-metrics.test.ts new file mode 100644 index 0000000000..d8227cb983 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/workspace-fleet-metrics.test.ts @@ -0,0 +1,246 @@ +import { describe, it, expect } from 'vitest' +import { applyCNPGFleetMetrics, buildCNPGFleet, CNPG_WORKSPACE_KEYS, type CNPGFleetMetricsReading, type CNPGWorkspaceResponse } from './workspace' + +function pod(cluster: string, n: number, role: 'primary' | 'replica'): any { + return { + metadata: { name: `${cluster}-${n}`, namespace: 'db', labels: { 'cnpg.io/cluster': cluster, 'cnpg.io/instanceRole': role } }, + status: { conditions: [{ type: 'Ready', status: 'True' }] }, + } +} + +function fleet() { + const coverage: CNPGWorkspaceResponse['coverage'] = {} + for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } + const cluster = (name: string, instances: number) => ({ + apiVersion: 'postgresql.cnpg.io/v1', + kind: 'Cluster', + metadata: { name, namespace: 'db' }, + spec: { instances }, + status: { phase: 'Cluster in healthy state', readyInstances: instances, currentPrimary: `${name}-1` }, + }) + return buildCNPGFleet({ + installed: true, + context: 'test', + namespaces: null, + coverage, + objects: { + clusters: [cluster('ha', 2), cluster('solo', 1), cluster('dark', 2)], + pods: [pod('ha', 1, 'primary'), pod('ha', 2, 'replica'), pod('solo', 1, 'primary'), pod('dark', 1, 'primary'), pod('dark', 2, 'replica')], + }, + issues: [], + audit: [], + backupsOmitted: 0, + }) +} + +const reading = (name: string, lag: CNPGFleetMetricsReading['lag'], growth: CNPGFleetMetricsReading['growth'] = { state: 'noSeries' }): CNPGFleetMetricsReading => ({ + namespace: 'db', + name, + lag, + growth, +}) + +const row = (f: ReturnType, name: string) => f.rows.find((r) => r.name === name)! + +describe('applyCNPGFleetMetrics', () => { + it('replaces "lag unknown" with the measured standby lag and names its source', () => { + const f = applyCNPGFleetMetrics(fleet(), [reading('ha', { state: 'ok', seconds: 7.2, pod: 'ha-2' })], { source: 'prometheus', lagSource: 'Prometheus cnpg_pg_replication_lag' }) + const r = row(f, 'ha') + expect(r.replication.text).toBe('1/1 Pods ready · lag 7.2 s') + expect(r.replication.tone).toBe('degraded') + expect(r.replication.source).toContain('ha-2') + expect(r.replication.source).toContain('cnpg_pg_replication_lag') + }) + + it('says why lag is unknown instead of zero, and leaves non-replica facts alone', () => { + const f = applyCNPGFleetMetrics(fleet(), [reading('dark', { state: 'noSeries', reason: 'no exporter series' })], { source: 'prometheus' }) + expect(row(f, 'dark').replication.text).toBe('1/1 Pods ready · lag unknown') + expect(row(f, 'dark').replication.source).toBeTruthy() + expect(row(f, 'dark').replication.tone).toBe('unknown') + expect(row(f, 'solo').replication.text).toBe('Single instance') + + const denied = applyCNPGFleetMetrics(fleet(), [reading('ha', { state: 'denied', grant: { verb: 'get', resource: 'pods', namespace: 'db' } })], { source: 'prometheus' }) + expect(row(denied, 'ha').replication.text).toBe('1/1 Pods ready · lag unknown') + expect(row(denied, 'ha').replication.source).toBe('Needs get pods in namespace db') + + const none = applyCNPGFleetMetrics(fleet(), undefined, { source: 'none', reason: 'Radar is not connected to Prometheus' }) + expect(row(none, 'ha').replication.text).toBe('1/1 Pods ready · lag unknown') + expect(row(none, 'ha').replication.source).toBe('Prometheus not connected') + expect(row(none, 'ha').replication.detail).toBe('Radar is not connected to Prometheus') + expect(row(none, 'ha').diskGrowth).toBeUndefined() + }) + + it('keeps the fleet untouched when no reading was requested', () => { + const base = fleet() + expect(applyCNPGFleetMetrics(base, undefined, undefined)).toBe(base) + expect(row(base, 'ha').replication.text).toBe('1/1 Pods ready · lag unknown') + }) + + it('reports disk growth only when measured', () => { + const f = applyCNPGFleetMetrics( + fleet(), + [reading('ha', { state: 'ok', seconds: 0.2, pod: 'ha-2' }, { state: 'ok', bytesPerHour: 1024 ** 3 / 24, claim: 'ha-1', instance: 'ha-1' })], + { source: 'prometheus', growthSource: 'deriv' }, + ) + expect(row(f, 'ha').diskGrowth?.text).toBe('+1.0 GiB/day') + expect(row(f, 'ha').replication.tone).toBe('healthy') + expect(row(f, 'dark').diskGrowth).toBeUndefined() + }) +}) + +describe('sustained replication lag', () => { + const src = { source: 'prometheus' as const, lagSource: 'Prometheus cnpg_pg_replication_lag' } + + it('puts a cluster into Needs attention only when lag stayed high for the whole window', () => { + const f = applyCNPGFleetMetrics( + fleet(), + [ + reading('ha', { state: 'ok', seconds: 95, pod: 'ha-2', sustainedSeconds: 40, sustainedPod: 'ha-2', sustainedWindow: '10m0s' }), + reading('dark', { state: 'ok', seconds: 120, pod: 'dark-2' }), + ], + src, + ) + const ha = row(f, 'ha') + expect(ha.attention).toBe(true) + expect(ha.problems[0]).toMatchObject({ severity: 'warning', category: 'availability', source: 'measurement' }) + expect(ha.problems[0].title).toBe('ha-2 ≥ 40 s behind in every sample for 10 min') + expect(ha.problems[0]).toMatchObject({ measuredBy: 'Prometheus' }) + expect(ha.problems[0].detail).not.toMatch(/cnpg_/) + expect(ha.problems[0].detail).toContain("If Prometheus missed some scrapes, those moments aren't included") + expect(ha.problems[0].shortTitle).toBe('ha-2: all samples ≥ 40 s behind (10 min)') + expect(ha.problems[0].detail).not.toContain('whole window') + expect(row(f, 'dark').attention).toBe(false) + expect(f.attentionCount).toBe(1) + }) + + it('says when the series were matched to the cluster by Pod name only', () => { + const note = "Matched by namespace and Pod names. Radar couldn't confirm these series belong to this exact cluster (no cluster label it could check)" + const f = applyCNPGFleetMetrics( + fleet(), + [reading('ha', { state: 'ok', seconds: 95, pod: 'ha-2', sustainedSeconds: 40, sustainedPod: 'ha-2', sustainedWindow: '10m0s', isolation: { mode: 'unverified', note } })], + src, + ) + const p = row(f, 'ha').problems[0] + expect(p.measuredBy).toBe('Prometheus, matched by Pod name') + expect(p.unverifiedMatch).toBe(true) + expect(p.detail).toContain(note) + const verified = applyCNPGFleetMetrics( + fleet(), + [reading('ha', { state: 'ok', seconds: 95, pod: 'ha-2', sustainedSeconds: 40, sustainedPod: 'ha-2', sustainedWindow: '10m0s', isolation: { mode: 'verified', note: 'x' } })], + src, + ) + expect(row(verified, 'ha').problems[0].measuredBy).toBe('Prometheus') + }) + + it('escalates to critical past five minutes and ignores a floor under 30 s', () => { + const f = applyCNPGFleetMetrics( + fleet(), + [ + reading('ha', { state: 'ok', seconds: 400, pod: 'ha-2', sustainedSeconds: 360, sustainedPod: 'ha-2', sustainedWindow: '10m0s' }), + reading('dark', { state: 'ok', seconds: 20, pod: 'dark-2', sustainedSeconds: 12, sustainedPod: 'dark-2', sustainedWindow: '10m0s' }), + ], + src, + ) + expect(row(f, 'ha').problems[0]).toMatchObject({ severity: 'critical', title: 'ha-2 ≥ 6 min behind in every sample for 10 min' }) + expect(row(f, 'dark').problems).toHaveLength(0) + }) +}) + +describe('standbys that receive nothing, and the WAL slots hold', () => { + const src = { source: 'prometheus' as const, lagSource: 'Prometheus cnpg_pg_replication_lag' } + it('a standby whose WAL receiver is down is a problem, never "lag 0 s"', () => { + const f = applyCNPGFleetMetrics( + fleet(), + [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-2', standbys: 1, receiving: 0, receiverDown: ['ha-2'], receiverDownSustained: ['ha-2'], receiverDownWindow: '5m0s' })], + src, + ) + const r = row(f, 'ha') + expect(r.replication.text).toBe('1/1 Pods ready · ha-2 not receiving WAL') + expect(r.replication.tone).toBe('unhealthy') + expect(r.attention).toBe(true) + const p = r.problems.find((x) => x.id === 'standby:db/ha:ha-2')! + expect(p).toMatchObject({ severity: 'critical', title: 'ha-2 is not receiving WAL from the primary', subject: { kind: 'Pod', name: 'ha-2' }, measuredBy: 'Prometheus' }) + expect(p.detail).toContain('Its WAL receiver was down in every sample Prometheus recorded over the last 5 minutes') + }) + it('a receiver down for less than the window is shown, not raised: a restarting standby reconnects on its own', () => { + const f = applyCNPGFleetMetrics(fleet(), [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-2', standbys: 1, receiving: 0, receiverDown: ['ha-2'], receiverDownSustained: [], receiverDownWindow: '5m0s' })], src) + expect(row(f, 'ha').replication.text).toBe('1/1 Pods ready · ha-2 not receiving WAL') + expect(row(f, 'ha').problems.some((p) => p.id.startsWith('standby:'))).toBe(false) + }) + it('lag alone does not establish streaming when the exporter reports no receiver state', () => { + const f = applyCNPGFleetMetrics(fleet(), [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-2', standbys: 1, receiverUnknown: true })], src) + expect(row(f, 'ha').replication).toMatchObject({ text: '1/1 Pods ready · lag 0 s · streaming unverified', tone: 'unknown' }) + expect(row(f, 'ha').attention).toBe(false) + }) + it('raises an inactive slot only from the primary, at 1 GiB or more, naming the standby it serves', () => { + const big = 4.9e9 + const f = applyCNPGFleetMetrics( + fleet(), + [ + { + ...reading('ha', { state: 'ok', seconds: 0.1, pod: 'ha-2', standbys: 1, receiving: 1 }), + slots: { + state: 'ok', + inactive: [ + { slot: '_cnpg_ha_2', pod: 'ha-1', role: 'primary', bytes: big }, + { slot: '_cnpg_ha_2', pod: 'ha-2', role: 'standby', bytes: big }, + ], + }, + }, + { ...reading('dark', { state: 'ok', seconds: 0.1, pod: 'dark-2', standbys: 1, receiving: 1 }), slots: { state: 'ok', inactive: [{ slot: '_cnpg_dark_9', pod: 'dark-2', role: 'standby', bytes: big }] } }, + ], + src, + ) + const slots = row(f, 'ha').problems.filter((p) => p.id.startsWith('slot:')) + expect(slots).toHaveLength(1) + expect(slots[0].title).toBe('Inactive slot _cnpg_ha_2 holds 4.6 GiB of WAL on ha-1 for ha-2') + expect(row(f, 'dark').problems.some((p) => p.id.startsWith('slot:'))).toBe(false) + const small = applyCNPGFleetMetrics(fleet(), [{ ...reading('ha', { state: 'ok', seconds: 0.1, pod: 'ha-2' }), slots: { state: 'ok', inactive: [{ slot: '_cnpg_ha_2', pod: 'ha-1', role: 'primary', bytes: 5e8 }] } }], src) + expect(row(small, 'ha').problems.some((p) => p.id.startsWith('slot:'))).toBe(false) + }) +}) + +describe('none receiving needs every expected standby accounted for', () => { + it('one standby down with another unreported is a warning, not "none receive"', () => { + const f = applyCNPGFleetMetrics( + fleet(), + [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-2', standbys: 1, receiving: 0, receiverDown: ['ha-2'], receiverDownSustained: ['ha-2'], receiverDownWindow: '5m0s' })], + { source: 'prometheus' }, + ) + expect(row(f, 'ha').problems.find((p) => p.id === 'standby:db/ha:ha-2')?.severity).toBe('critical') + const three = applyCNPGFleetMetrics( + fleet(), + [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-2', standbys: 2, receiving: 0, receiverDown: ['ha-2'], receiverDownSustained: ['ha-2'], receiverDownWindow: '5m0s' })], + { source: 'prometheus' }, + ) + // spec.instances 2 expects one standby; two reporting with one down is not "all down". + expect(row(three, 'ha').problems.find((p) => p.id === 'standby:db/ha:ha-2')?.severity).toBe('warning') + }) +}) + +describe('fleet lag covers the standbys that report', () => { + it('says how many standbys the lag covers, and is not healthy while one is unreported', () => { + const three = fleet() + three.rows.find((r) => r.name === 'ha')!.instances.desired = 3 + const f = applyCNPGFleetMetrics(three, [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-2', lagStandbys: 1, standbys: 1, receiving: 1 })], { source: 'prometheus' }) + expect(row(f, 'ha').replication).toMatchObject({ text: expect.stringContaining('lag 0 s (1 of 2 standbys reporting)'), tone: 'unknown' }) + }) + + it('counts the standbys whose lag was read, not those whose receiver was', () => { + const three = fleet() + three.rows.find((r) => r.name === 'ha')!.instances.desired = 3 + const f = applyCNPGFleetMetrics(three, [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-2', lagStandbys: 1, standbys: 2, receiving: 2 })], { source: 'prometheus' }) + expect(row(f, 'ha').replication).toMatchObject({ text: expect.stringContaining('(1 of 2 standbys reporting)'), tone: 'unknown' }) + }) + + it('expects every instance of a replica cluster to report, its designated primary included', () => { + const replica = fleet() + const r = replica.rows.find((x) => x.name === 'ha')! + r.instances.desired = 3 + r.replicaCluster = { source: 'pg-origin' } + const f = applyCNPGFleetMetrics(replica, [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-1', lagStandbys: 2, standbys: 2, receiving: 2 })], { source: 'prometheus' }) + expect(row(f, 'ha').replication).toMatchObject({ text: expect.stringContaining('(2 of 3 instances reporting)'), tone: 'unknown' }) + const all = applyCNPGFleetMetrics(replica, [reading('ha', { state: 'ok', seconds: 0, pod: 'ha-1', lagStandbys: 3, standbys: 3, receiving: 3 })], { source: 'prometheus' }) + expect(row(all, 'ha').replication.text).not.toContain('reporting') + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/workspace-problems.test.ts b/packages/k8s-ui/src/components/cnpg/workspace-problems.test.ts new file mode 100644 index 0000000000..d7f97431c8 --- /dev/null +++ b/packages/k8s-ui/src/components/cnpg/workspace-problems.test.ts @@ -0,0 +1,175 @@ +import { describe, expect, it } from 'vitest' +import { cnpgCollapseBackupFailures, cnpgCompareProblems, cnpgFoldLastBackupFailed, cnpgFormatLag, cnpgIssueOrigin, cnpgIssueText, cnpgIssueTitle, type CNPGProblem } from './workspace' + +const problem = (title: string, severity: CNPGProblem['severity'], kind: string, group = ''): CNPGProblem => ({ + id: title, + severity, + category: kind === 'Pod' ? 'availability' : 'protection', + title, + subject: { kind, group, namespace: 'pg', name: kind === 'Pod' ? 'pg-wal-failing-1' : 'pg-wal-failing' }, + source: 'issue', +}) + +describe('cnpgCompareProblems', () => { + it('puts the cluster-level cause ahead of a Pod symptom at the same severity (pg-wal-failing)', () => { + const probe = problem('pg-wal-failing-1 not ready (readiness probe failing)', 'critical', 'Pod') + const wal = problem('WAL archiving failing', 'critical', 'Cluster', 'postgresql.cnpg.io') + expect([probe, wal].sort(cnpgCompareProblems).map((p) => p.title)).toEqual(['WAL archiving failing', 'pg-wal-failing-1 not ready (readiness probe failing)']) + }) + it('keeps severity first', () => { + const crit = problem('pg-1 restarted recently (CrashLoopBackOff)', 'critical', 'Pod') + const warn = problem('Backup failed', 'warning', 'Backup', 'postgresql.cnpg.io') + expect([warn, crit].sort(cnpgCompareProblems)[0]).toBe(crit) + }) +}) + +describe('cnpgIssueTitle', () => { + it('turns a bare reason into a sentence about the subject', () => { + expect(cnpgIssueTitle({ kind: 'Pod', name: 'pg-wal-failing-1', reason: 'ReadinessProbeFailed', message: 'ReadinessProbeFailed' })).toBe( + 'pg-wal-failing-1 not ready (readiness probe failing)', + ) + expect(cnpgIssueTitle({ kind: 'Pod', name: 'pg-runtime-5', reason: 'CrashLoopBackOff' })).toBe('pg-runtime-5 restarted recently (CrashLoopBackOff)') + expect(cnpgIssueTitle({ kind: 'Backup', name: 'b-1', reason: 'BackupStuck' })).toBe('Backup b-1: backup stuck') + }) + it('keeps a real message', () => { + expect(cnpgIssueTitle({ kind: 'Cluster', name: 'pg', reason: 'ContinuousArchivingFailing', message: 'WAL archiving failing' })).toBe('WAL archiving failing') + }) +}) + +describe('cnpgIssueText', () => { + it('heads a CNPG condition issue with a plain title and keeps the operator message beneath', () => { + const t = cnpgIssueText({ + kind: 'Cluster', + name: 'pg-wal-failing', + reason: 'CNPGWALArchivingFailing', + message: + 'The last WAL archival did not complete; recovery-point advancement is uncertain: unexpected failure invoking barman-cloud-wal-archive: exit status 4', + }) + expect(t.title).toBe('WAL archiving failing') + expect(t.detail).toContain('exit status 4') + }) +}) + +describe('cnpgIssueText for certificates and schedules', () => { + it('words expired and expiring certificates apart, and does not claim no backup was produced', () => { + expect(cnpgIssueText({ kind: 'Cluster', name: 'pg', reason: 'CNPGCertificateExpired', message: 'The certificate in Secret pg-ca expired 2026-09-30T11:00:00Z' }).title).toBe( + 'A certificate has expired', + ) + expect(cnpgIssueText({ kind: 'Cluster', name: 'pg', reason: 'CNPGCertificateExpiring', message: 'The certificate in Secret x expires in 3 days' }).title).toBe( + 'A certificate expires soon', + ) + expect(cnpgIssueText({ kind: 'ScheduledBackup', name: 's', reason: 'CNPGScheduledRunNoBackup', message: 'x y' }).title).toBe('No successful backup since a scheduled run') + }) +}) + +describe('backup failures', () => { + it('does not repeat the title at the start of the detail', () => { + expect(cnpgIssueText({ kind: 'Backup', name: 'b', reason: 'CNPGBackupFailed', message: 'Backup failed: cannot proceed with the backup' })).toEqual({ + title: 'Backup failed', + detail: 'Cannot proceed with the backup', + }) + }) + it('collapses Backups that failed the same way into one problem about the latest, by the Backups\' own times', () => { + // detectCNPGBackupIssues emits no first_seen: the latest comes from the Backup objects. + const problems = ['b-1', 'b-3', 'b-2'].map((n) => { + const t = cnpgIssueText({ kind: 'Backup', name: n, reason: 'CNPGBackupFailed', message: 'Backup failed: cannot proceed as the cluster has no plugin configured' }) + return { + id: n, + severity: 'warning' as const, + category: 'protection' as const, + ...t, + subject: { kind: 'Backup', group: 'postgresql.cnpg.io', namespace: 'pg', name: n }, + source: 'issue' as const, + reason: 'CNPGBackupFailed', + } + }) + const times = new Map([ + ['b-1', Date.parse('2026-09-20T00:00:00Z')], + ['b-3', Date.parse('2026-09-21T00:00:00Z')], + ['b-2', Date.parse('2026-09-22T00:00:00Z')], + ]) + const out = cnpgCollapseBackupFailures(problems, times) + expect(out).toHaveLength(1) + expect(out[0]).toMatchObject({ title: '3 backups failed: cannot proceed as the cluster has no plugin configured', subject: { name: 'b-2' } }) + expect(out[0].alsoAbout?.map((o) => o.name)).toEqual(['b-3', 'b-1']) + }) +}) + +describe('scheduled run without a backup', () => { + it('words the schedule and dates the run as an age from first_seen', () => { + const t = cnpgIssueText({ + kind: 'Cluster', + name: 'pg', + reason: 'CNPGScheduledRunNoBackup', + message: 'ScheduledBackup pg-nightly (every day at 02:00 UTC) has had no successful backup since its run', + first_seen: new Date(Date.now() - 2 * 24 * 3600 * 1000 - 60_000).toISOString(), + }) + expect(t.title).toBe('No successful backup since a scheduled run') + expect(t.detail).toBe('ScheduledBackup pg-nightly (every day at 02:00 UTC) has had no successful backup since its run 2d ago') + }) +}) + +describe('cnpgFormatLag', () => { + it('reads every replay lag the same way, rounded down', () => { + expect(cnpgFormatLag(0)).toBe('0 s') + expect(cnpgFormatLag(0.25)).toBe('250 ms') + expect(cnpgFormatLag(0.2509)).toBe('250 ms') + expect(cnpgFormatLag(8.27)).toBe('8.2 s') + expect(cnpgFormatLag(55.9)).toBe('55 s') + expect(cnpgFormatLag(1500.4)).toBe('25 min') + expect(cnpgFormatLag(1721.3)).toBe('28 min') + expect(cnpgFormatLag(3600)).toBe('1 h') + expect(cnpgFormatLag(4000)).toBe('1 h 6 min') + }) +}) + +describe('cnpgFoldLastBackupFailed', () => { + const p = (id: string, kind: string, name: string, reason: string, alsoAbout?: { kind: string; name: string }[]) => ({ + id, + severity: 'warning' as const, + category: 'protection' as const, + title: id, + subject: { kind, group: 'postgresql.cnpg.io', namespace: 'pg', name }, + source: 'issue' as const, + reason, + alsoAbout, + }) + const last = p('last', 'Cluster', 'pg', 'CNPGLastBackupFailed') + it('drops "the last backup failed" when a failed-Backup problem already covers the newest Backup', () => { + const group = p('3 backups failed', 'Backup', 'b-3', 'CNPGBackupFailed', [{ kind: 'Backup', name: 'b-2' }]) + const out = cnpgFoldLastBackupFailed([group, last], 'b-3') + expect(out.map((x) => x.id)).toEqual(['3 backups failed']) + expect(out[0].reason).toBe('CNPGBackupFailed') + }) + it('counts only a failed-Backup problem as covering the newest Backup', () => { + const other = p('stuck', 'Backup', 'b-3', 'CNPGBackupStuck') + expect(cnpgFoldLastBackupFailed([other, last], 'b-3').map((x) => x.id)).toEqual(['stuck', 'last']) + }) + it('keeps it when the newest Backup is not among the failures, or is unknown', () => { + const old = p('old failure', 'Backup', 'b-1', 'CNPGBackupFailed') + expect(cnpgFoldLastBackupFailed([old, last], 'b-9')).toHaveLength(2) + expect(cnpgFoldLastBackupFailed([old, last], undefined)).toHaveLength(2) + }) +}) + +describe('cnpgIssueOrigin', () => { + it('names where the evidence comes from', () => { + expect(cnpgIssueOrigin({ kind: 'Cluster', reason: 'CNPGWALArchivingFailing' })).toEqual({ label: 'Reported by CNPG', detail: 'Cluster ContinuousArchiving condition' }) + expect(cnpgIssueOrigin({ kind: 'Backup', reason: 'CNPGWALArchivingFailing' }).label).toBe('Backup status') + expect(cnpgIssueOrigin({ kind: 'Cluster', reason: 'CNPGClusterDegraded' }).label).toBe('Radar check of ready instances') + expect(cnpgIssueOrigin({ kind: 'Cluster', reason: 'CNPGLastBackupFailed' }).label).toBe('Reported by CNPG') + expect(cnpgIssueOrigin({ kind: 'Database', reason: 'CNPGDeclarativeNotApplied' }).label).toBe('Reported by CNPG') + expect(cnpgIssueOrigin({ kind: 'ScheduledBackup', reason: 'CNPGScheduledBackupMissed' }).label).toBe('Radar check of the backup schedule') + expect(cnpgIssueOrigin({ kind: 'Pod', reason: 'HighRestartCount' }).label).toBe('Radar check of restarts') + expect(cnpgIssueOrigin({ kind: 'Pod', reason: 'ReadinessProbeInvalid' }).label).toBe('Radar check of the probe') + expect(cnpgIssueOrigin({ kind: 'Cluster', reason: 'CNPGClusterFailingOver' }).label).toBe('Reported by CNPG') + expect(cnpgIssueOrigin({ kind: 'Backup', reason: 'CNPGBackupFailed' }).label).toBe('Backup status') + expect(cnpgIssueOrigin({ kind: 'Cluster', reason: 'CNPGScheduledRunNoBackup' }).label).toBe('Radar check of the backup schedule') + expect(cnpgIssueOrigin({ kind: 'Cluster', reason: 'CNPGCertificateExpired' }).label).toBe('Certificate expiry (from Cluster status)') + expect(cnpgIssueOrigin({ kind: 'Pod', reason: 'ReadinessProbeFailed' }).label).toBe('Pod readiness probe') + expect(cnpgIssueOrigin({ kind: 'Pod', reason: 'CrashLoopBackOff' }).label).toBe('Pod status') + }) + it('falls back to "Detected by Radar" without inventing a source', () => { + expect(cnpgIssueOrigin({ kind: 'Pooler', reason: 'SomethingNew' })).toEqual({ label: 'Detected by Radar' }) + }) +}) diff --git a/packages/k8s-ui/src/components/cnpg/workspace.test.ts b/packages/k8s-ui/src/components/cnpg/workspace.test.ts index beeb5dd65b..34c4092485 100644 --- a/packages/k8s-ui/src/components/cnpg/workspace.test.ts +++ b/packages/k8s-ui/src/components/cnpg/workspace.test.ts @@ -1,8 +1,29 @@ +import { cnpgDimensions } from './ha' import { describe, it, expect } from 'vitest' -import { buildCNPGFleet, type CNPGWorkspaceResponse, type CNPGWorkspaceKey, CNPG_WORKSPACE_KEYS } from './workspace' +import { buildCNPGFleet, cnpgReadyInstances, cnpgRecoveryMatchesCluster, getCNPGRestoreValidation, type CNPGWorkspaceResponse, type CNPGWorkspaceKey, CNPG_WORKSPACE_KEYS } from './workspace' const G = 'postgresql.cnpg.io/v1' +it('does not promote an archiving condition into evidence of an archive destination', () => { + const c = cluster('payments', 'db', { status: { conditions: [{ type: 'ContinuousArchiving', status: 'True', lastTransitionTime: '2026-10-01T12:00:00Z' }] } }) + const fact = buildCNPGFleet(resp({ clusters: [c] })).rows[0].protection.walArchiving + expect(fact).toMatchObject({ text: 'Not archived: no destination configured', tone: 'neutral', source: 'Cluster spec' }) + expect(fact.detail).toContain("WAL is not archived to recovery storage") + expect(fact.operatorCondition).toMatchObject({ type: 'ContinuousArchiving', status: 'True', lastTransitionTime: '2026-10-01T12:00:00Z' }) + const snapshot = cluster('snapshot', 'db', { spec: { backup: { volumeSnapshot: {} } }, status: c.status }) + expect(buildCNPGFleet(resp({ clusters: [snapshot] })).rows[0].protection.walArchiving.text).toBe('Not archived: no destination configured') +}) + +it('preserves declared archiver failures and distinguishes backup-only and opaque archiver plugins', () => { + const make = (plugins: any[], status = 'True') => buildCNPGFleet(resp({ clusters: [cluster('pg', 'db', { spec: { plugins }, status: { conditions: [{ type: 'ContinuousArchiving', status, message: 'archive report' }] } })] })).rows[0].protection.walArchiving + const barman = { name: 'barman-cloud.cloudnative-pg.io', isWALArchiver: true } + expect(make([barman], 'False')).toMatchObject({ text: 'Failing', tone: 'unhealthy', detail: 'archive report' }) + expect(make([barman])).toMatchObject({ text: 'Not archived: no destination configured', tone: 'neutral' }) + expect(make([{ ...barman, isWALArchiver: false, parameters: { barmanObjectName: 'store' } }])).toMatchObject({ text: 'Not archived: no destination configured', tone: 'neutral' }) + expect(make([{ name: 'third-party-archive', isWALArchiver: true }])).toMatchObject({ text: 'CNPG reports archiving', detail: 'Archive plugin declared; its destination is not assessed here' }) + expect(make([{ ...barman, enabled: false, parameters: { barmanObjectName: 'store' } }])).toMatchObject({ text: 'Not archived: no destination configured', tone: 'neutral' }) +}) + function cluster(name: string, ns: string, extra: any = {}): any { return { apiVersion: G, @@ -27,6 +48,10 @@ function pod(name: string, ns: string, clusterName: string, role: string, ready } } +function serverProblem(reason: string, message: string, kind = 'Cluster', name = 'pg-a', severity: 'critical' | 'warning' = 'warning') { + return { id: `server:${reason}`, reason, message, kind, name, namespace: 'db', group: 'postgresql.cnpg.io', severity } +} + function resp(objects: Partial>, over: Partial = {}): CNPGWorkspaceResponse { const coverage: CNPGWorkspaceResponse['coverage'] = {} for (const k of CNPG_WORKSPACE_KEYS) coverage[k] = { state: 'full' } @@ -39,6 +64,7 @@ function resp(objects: Partial>, over: Partial { ) const row = fleet.rows[0] expect(row.replication.tone).toBe('unknown') - expect(row.replication.text).toContain('2/2 replicas ready') + expect(row.replication.text).toContain('2/2 Pods ready') expect(row.pods[0].role).toBe('primary') }) @@ -81,6 +107,22 @@ describe('buildCNPGFleet', () => { expect(fleet.rows[0].name).toBe('pg-b') }) + it('orders rows by worst problem, then problem count, then namespace/name', () => { + const issue = (id: string, severity: 'critical' | 'warning', name: string) => ({ + id, severity, kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: 'db', name, reason: 'CNPGClusterUnhealthy', message: id, + }) + const fleet = buildCNPGFleet( + resp( + { clusters: ['pg-a', 'pg-b', 'pg-c', 'pg-d', 'pg-e'].map((n) => cluster(n, 'db')) }, + { + issues: [issue('w1', 'warning', 'pg-a'), issue('c1', 'critical', 'pg-b'), issue('w2', 'warning', 'pg-c'), issue('w3', 'warning', 'pg-c')], + audit: [{ checkId: 'cnpgNoDeclarativeBackup', severity: 'warning', kind: 'Cluster', namespace: 'db', name: 'pg-e', message: 'no ScheduledBackup' }], + }, + ), + ) + expect(fleet.rows.map((r) => r.name)).toEqual(['pg-b', 'pg-c', 'pg-a', 'pg-e', 'pg-d']) + }) + it('treats the no-schedule audit finding as posture, not attention, and words it narrowly', () => { const fleet = buildCNPGFleet( resp({ clusters: [cluster('pg-a', 'db')] }, { @@ -131,7 +173,7 @@ describe('buildCNPGFleet', () => { })], }), ) - expect(fleet.rows[0].protection.lastSuccessfulBackup.text).toBe('None observed') + expect(fleet.rows[0].protection.lastSuccessfulBackup.text).toBe('No successful backup yet') }) it('never marks restore validation healthy', () => { @@ -144,15 +186,49 @@ describe('buildCNPGFleet', () => { }) const fleet = buildCNPGFleet(resp({ clusters: [src, restored] })) const a = fleet.rows.find((r) => r.name === 'pg-a')! - expect(a.protection.restoreValidation.text).toBe('Restored into pg-a-restore') + expect(a.protection.restoreValidation.text).toBe('Archive restored into pg-a-restore') expect(a.protection.restoreValidation.tone).toBe('neutral') const r = fleet.rows.find((x) => x.name === 'pg-a-restore')! expect(r.protection.restoreValidation.text).toBe('None recorded') expect(r.protection.restoreValidation.tone).toBe('unknown') }) - it('ends the recovery window at WAL archiving, not at the last base backup', () => { + it('matches a restored Cluster’s external source despite conflicting Backup parameters', () => { + const src = cluster('pg-a', 'db', { spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'cluster-store', serverName: 'cluster-server' } }] } }) + const conflicting = { apiVersion: 'postgresql.cnpg.io/v1', kind: 'Backup', metadata: { name: 'b', namespace: 'db' }, spec: { cluster: { name: 'pg-a' }, method: 'plugin', pluginConfiguration: { name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'backup-store', serverName: 'backup-server' } } }, status: { phase: 'completed' } } + const restored = cluster('pg-a-restore', 'db', { spec: { bootstrap: { recovery: { source: 'origin' } }, externalClusters: [{ name: 'origin', plugin: { name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'cluster-store', serverName: 'cluster-server' } } }] } }) + const fact = buildCNPGFleet(resp({ clusters: [src, restored], backups: [conflicting] })).rows.find((r) => r.name === 'pg-a')!.protection.restoreValidation + expect(fact).toMatchObject({ text: 'Archive restored into pg-a-restore', tone: 'neutral' }) + const wrongSource = { ...restored, spec: { ...restored.spec, externalClusters: [{ name: 'origin', plugin: { name: 'barman-cloud.cloudnative-pg.io', parameters: conflicting.spec.pluginConfiguration.parameters } }] } } + expect(buildCNPGFleet(resp({ clusters: [src, wrongSource], backups: [conflicting] })).rows.find((r) => r.name === 'pg-a')!.protection.restoreValidation).toMatchObject({ text: 'None recorded', tone: 'unknown' }) + }) + + it('reads a recorded validation note on the restored cluster, still never healthy', () => { const plugin = { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store' } }] } + const src = cluster('pg-a', 'db', { metadata: { uid: 'uid-a' }, spec: plugin }) + const restoredSpec = { + bootstrap: { recovery: { source: 'origin' } }, + externalClusters: [{ name: 'origin', plugin: { name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store', serverName: 'pg-a' } } }], + } + const note = (uid: string) => + JSON.stringify({ version: 1, recordedAt: '2026-09-29T10:00:00Z', recordedBy: 'alice', checked: 'row counts on orders', source: { namespace: 'db', name: 'pg-a', uid, verified: true } }) + const noted = cluster('pg-a-restore', 'db', { metadata: { annotations: { 'radar.skyhook.io/restore-validation': note('uid-a') } }, spec: restoredSpec }) + const fleet = buildCNPGFleet(resp({ clusters: [src, noted] })) + const fact = fleet.rows.find((r) => r.name === 'pg-a')!.protection.restoreValidation + expect(fact.text).toBe('Validation recorded') + expect(fact.tone).toBe('neutral') + expect(fact.at).toBe('2026-09-29T10:00:00Z') + expect(fact.source).toContain('by alice on pg-a-restore') + + const otherUID = cluster('pg-a-restore', 'db', { metadata: { annotations: { 'radar.skyhook.io/restore-validation': note('uid-previous-incarnation') } }, spec: restoredSpec }) + expect(buildCNPGFleet(resp({ clusters: [src, otherUID] })).rows.find((r) => r.name === 'pg-a')!.protection.restoreValidation.text).toBe('None recorded') + + const malformed = cluster('pg-a-restore', 'db', { metadata: { annotations: { 'radar.skyhook.io/restore-validation': '{not json' } }, spec: restoredSpec }) + expect(buildCNPGFleet(resp({ clusters: [src, malformed] })).rows.find((r) => r.name === 'pg-a')!.protection.restoreValidation.text).toBe('Archive restored into pg-a-restore') + }) + + it('ends the recovery window at WAL archiving, not at the last base backup', () => { + const plugin = { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', isWALArchiver: true, parameters: { barmanObjectName: 'store' } }] } const store = { apiVersion: 'barmancloud.cnpg.io/v1', kind: 'ObjectStore', metadata: { name: 'store', namespace: 'db' }, status: { serverRecoveryWindow: { 'pg-a': { firstRecoverabilityPoint: '2026-09-01T00:00:00Z', lastSuccessfulBackupTime: '2026-09-02T00:00:00Z', lastFailedBackupTime: '2026-09-03T00:00:00Z' } } }, @@ -165,9 +241,17 @@ describe('buildCNPGFleet', () => { expect(buildCNPGFleet(resp({ clusters: [failing], objectStores: [store] })).rows[0].protection.recoveryWindow.tone).toBe('degraded') }) + it('qualifies archiving as the operator report with its transition time', () => { + const hourAgo = new Date(Date.now() - 3_600_000).toISOString() + const resumed = cluster('pg-a', 'db', { spec: { backup: { barmanObjectStore: { destinationPath: 's3://backups' } } }, metadata: { creationTimestamp: '2026-01-01T00:00:00Z' }, status: { conditions: [{ type: 'ContinuousArchiving', status: 'True', lastTransitionTime: hourAgo }] } }) + expect(buildCNPGFleet(resp({ clusters: [resumed] })).rows[0].protection.walArchiving).toMatchObject({ text: 'CNPG reports archiving', source: 'Cluster status · ContinuousArchiving=True', at: hourAgo, atMeaning: 'since' }) + const old = cluster('pg-a', 'db', { spec: { backup: { barmanObjectStore: { destinationPath: 's3://backups' } } }, metadata: { creationTimestamp: '2026-01-01T00:00:00Z' }, status: { conditions: [{ type: 'ContinuousArchiving', status: 'True', lastTransitionTime: '2026-01-01T00:02:00Z' }] } }) + expect(buildCNPGFleet(resp({ clusters: [old] })).rows[0].protection.walArchiving.at).toBe('2026-01-01T00:02:00Z') + }) + it('reports WAL archiving from the condition and unknown when absent', () => { - const failing = cluster('pg-a', 'db', { status: { conditions: [{ type: 'ContinuousArchiving', status: 'False', message: 'exit status 1' }] } }) - const silent = cluster('pg-b', 'db') + const failing = cluster('pg-a', 'db', { spec: { backup: { barmanObjectStore: { destinationPath: 's3://backups' } } }, status: { conditions: [{ type: 'ContinuousArchiving', status: 'False', message: 'exit status 1' }] } }) + const silent = cluster('pg-b', 'db', { spec: { backup: { barmanObjectStore: { destinationPath: 's3://backups' } } } }) const fleet = buildCNPGFleet(resp({ clusters: [failing, silent] })) expect(fleet.rows.find((r) => r.name === 'pg-a')!.protection.walArchiving.tone).toBe('unhealthy') expect(fleet.rows.find((r) => r.name === 'pg-a')!.protection.summary.text).toBe('WAL archiving failing') @@ -245,11 +329,45 @@ describe('buildCNPGFleet', () => { expect(fleet.rows[0].categories.has('availability')).toBe(true) }) + it("names a Cluster's own Job Pod problem by what the Job is for, as the cause, never as an instance", () => { + const joinPod = { + apiVersion: 'v1', + kind: 'Pod', + metadata: { name: 'pg-a-2-join-x1', namespace: 'db', labels: { 'cnpg.io/cluster': 'pg-a', 'cnpg.io/jobRole': 'join', 'cnpg.io/instanceName': 'pg-a-2' } }, + status: { phase: 'Pending' }, + } + const base = resp({ clusters: [cluster('pg-a', 'db', { status: { readyInstances: 1 } })], pods: [pod('pg-a-1', 'db', 'pg-a', 'primary')] }, { + issues: [ + { id: 'c1', severity: 'warning', category: 'operator_condition_failed', kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: 'db', name: 'pg-a', reason: 'Ready: ClusterIsNotReady', message: 'Cluster Is Not Ready' }, + { id: 'j1', severity: 'critical', category: 'unschedulable', kind: 'Pod', namespace: 'db', name: 'pg-a-2-join-x1', reason: 'Unschedulable', message: '2 node(s) insufficient pods (0/2 nodes available)' }, + ], + }) + const row = buildCNPGFleet({ ...base, jobPods: [joinPod] }).rows[0] + expect(row.problems[0].title).toBe("New standby pg-a-2: Can't be scheduled") + expect(row.problems[0].severity).toBe('warning') + expect(row.problems[0].detail).toBe('both nodes have reached their Pod limit') + expect(row.problems[0].rawDetail).toBe('2 node(s) insufficient pods (0/2 nodes available)') + expect(row.problems[0].origin?.label).toBe('Kubernetes scheduler') + expect(row.pods.map((p) => p.name)).toEqual(['pg-a-1']) + expect(buildCNPGFleet(base).rows[0].problems.some((p) => p.subject.name === 'pg-a-2-join-x1')).toBe(false) + }) + it('reads partial coverage by allowed namespaces and treats unnamed partial coverage as unknown', () => { const named = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'partial', allowedNamespaces: ['db'] } } })) - expect(named.rows[0].protection.lastSuccessfulBackup.text).toBe('None observed') + expect(named.rows[0].protection.lastSuccessfulBackup.text).toBe('No successful backup yet') const unnamed = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'partial' } } })) - expect(unnamed.rows[0].protection.lastSuccessfulBackup.text).toBe('No access to Backups') + expect(unnamed.rows[0].protection.lastSuccessfulBackup.text).toBe('Backups not read in db') + }) + + it('names the cause of a partial read only when the server names the namespace', () => { + const denied = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { backups: { state: 'partial', allowedNamespaces: ['other'], deniedNamespaces: ['db'] } } })) + expect(denied.rows[0].protection.lastSuccessfulBackup.text).toBe('No access to Backups') + const uncached = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { pods: { state: 'partial', allowedNamespaces: ['other'], uncachedNamespaces: ['db'] } } })) + expect(uncached.rows[0].replication.text).toBe('Radar does not cache Pods in db') + const none = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { pods: { state: 'uncached' } } })) + expect(none.rows[0].replication.text).toBe('Radar does not cache Pods') + const mixed = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { pods: { state: 'uncached', uncachedNamespaces: ['other'], deniedNamespaces: ['db'] } } })) + expect(mixed.rows[0].replication.text).toBe('No access to Pods') }) it('says no access instead of "no replica pods" when Pods are unreadable', () => { @@ -282,4 +400,237 @@ describe('buildCNPGFleet', () => { expect(p.lastSuccessfulBackup.text).toBe('Loading…') expect(p.recoveryWindow.text).toBe('Loading…') }) + + it('shows the Pods’ ready count and raises an availability problem when status claims more', () => { + const fleet = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db')], + pods: [pod('pg-a-1', 'db', 'pg-a', 'primary', false), pod('pg-a-2', 'db', 'pg-a', 'replica'), pod('pg-a-3', 'db', 'pg-a', 'replica', false)], + }, { issues: [serverProblem('CNPGInstanceReadinessMismatch', '2 of 3 instance Pods not ready, including the primary', 'Cluster', 'pg-a', 'critical')] }), + ) + const row = fleet.rows[0] + expect(row.podReadiness).toEqual({ ready: 1, total: 3 }) + expect(row.readinessContradicted).toBe(true) + expect(cnpgReadyInstances(row)).toMatchObject({ text: '1/3', tone: 'degraded' }) + expect(cnpgReadyInstances(row).note).toContain('CNPG status reports 3 ready') + const problem = row.problems[0] + expect(problem.severity).toBe('critical') + expect(problem.category).toBe('availability') + expect(problem.title).toBe('2 of 3 instance Pods not ready, including the primary') + expect(row.attention).toBe(true) + expect(fleet.attentionCount).toBe(1) + }) + + it('agrees with status when the Pods do, and never judges unreadable Pods', () => { + const agreeing = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db', { status: { readyInstances: 2 } })], + pods: [pod('pg-a-1', 'db', 'pg-a', 'primary'), pod('pg-a-2', 'db', 'pg-a', 'replica'), pod('pg-a-3', 'db', 'pg-a', 'replica', false)], + }), + ).rows[0] + expect(agreeing.readinessContradicted).toBeUndefined() + expect(cnpgReadyInstances(agreeing)).toEqual({ text: '2/3' }) + const denied = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')] }, { coverage: { pods: { state: 'denied' } } })).rows[0] + expect(denied.podReadiness).toBeUndefined() + expect(denied.problems).toEqual([]) + }) + + it('separates missing operator readiness from an observed Pod count', () => { + const instances = { ready: null, desired: 1 } + expect(cnpgReadyInstances({ instances, podReadiness: { ready: 0, total: 0 } })).toEqual({ + text: 'Not reported by the operator', podText: '0 of 1 instance Pods ready', + }) + expect(cnpgReadyInstances({ instances })).toEqual({ text: 'Not reported by the operator', podText: undefined }) + expect(cnpgReadyInstances({ instances: { ready: null, desired: null }, podReadiness: { ready: 1, total: 1 } }).podText).toBe('1 instance Pod ready') + expect(cnpgReadyInstances({ instances: { ready: 0, desired: 1 } })).toEqual({ text: '0/1' }) + }) + + it('names both primaries when status and the role label disagree', () => { + const row = buildCNPGFleet( + resp({ + clusters: [cluster('pg-a', 'db', { status: { currentPrimary: 'pg-a-2', readyInstances: 1 } })], + pods: [pod('pg-a-1', 'db', 'pg-a', 'primary'), pod('pg-a-2', 'db', 'pg-a', 'replica', false)], + }, { issues: [serverProblem('CNPGPrimaryLabelMismatch', 'CNPG status names pg-a-2 primary; the Pod labelled primary is pg-a-1')] }), + ).rows[0] + expect(row.primaryConflict).toEqual({ status: 'pg-a-2', labelled: 'pg-a-1' }) + expect(row.problems.map((p) => p.title)).toContain('CNPG status names pg-a-2 primary; the Pod labelled primary is pg-a-1') + }) +}) + +describe('schedule fact', () => { + it('reads the schedule in words when the server supplied a reading, with the cron as its source', () => { + const sched = { metadata: { namespace: 'db', name: 'nightly' }, spec: { cluster: { name: 'pg-a' }, schedule: '0 0 2 * * *' } } + const withReading = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')], scheduledBackups: [sched] }, { scheduleReadings: { 'db/nightly': 'every day at 02:00 UTC' } })) + expect(withReading.rows[0].protection.schedule).toMatchObject({ text: 'Enabled · every day at 02:00 UTC · blocked: no backup destination', source: 'ScheduledBackup nightly · cron 0 0 2 * * *' }) + const without = buildCNPGFleet(resp({ clusters: [cluster('pg-a', 'db')], scheduledBackups: [sched] })) + expect(without.rows[0].protection.schedule.text).toBe('Enabled · 0 0 2 * * * · blocked: no backup destination') + }) +}) + + describe('CNPG certainty', () => { + const archiving = cluster('pg', 'db', { spec: { backup: { barmanObjectStore: { destinationPath: 's3://backups' } } }, status: { conditions: [{ type: 'ContinuousArchiving', status: 'True' }] } }) + it('degrades an archiving cluster with no completed Backup and names the read', () => { + const p = buildCNPGFleet(resp({ clusters: [archiving] })).rows[0].protection + expect(p.lastSuccessfulBackup).toMatchObject({ tone: 'degraded', text: 'No successful backup yet', source: 'Backups read in this namespace; none completed' }) + }) + it('keeps unread Backups unknown', () => { + const p = buildCNPGFleet(resp({ clusters: [archiving] }, { coverage: { backups: { state: 'denied' } } })).rows[0].protection + expect(p.lastSuccessfulBackup).toMatchObject({ tone: 'unknown', source: 'Backups not read' }) + }) + const declaration = (applied: boolean, observed = 2) => ({ apiVersion: G, kind: 'Database', metadata: { name: 'app', namespace: 'db', generation: 2 }, spec: { cluster: { name: 'pg' } }, status: { applied, observedGeneration: observed } }) + it('shows a qualified lower bound inline when a declaration kind was not read', () => { + const d = buildCNPGFleet(resp({ clusters: [archiving], databases: [declaration(true)] }, { coverage: { publications: { state: 'denied' } } })).rows[0].declarations + expect(d.summary).toEqual({ text: '≥1 reconciled; Publications not read', tone: 'unknown' }) + }) + it('does not use an exact denominator when some declarations were not read', () => { + const d = buildCNPGFleet(resp({ clusters: [archiving], databases: [declaration(false)] }, { coverage: { subscriptions: { state: 'denied' } } })).rows[0].declarations + expect(d.summary.text).toBe('≥1 not reconciled; Subscriptions not read') + expect(d.summary.tone).toBe('degraded') + }) + it('counts stale success and failure as pending in the fleet', () => { + for (const applied of [true, false]) { + const d = buildCNPGFleet(resp({ clusters: [archiving], databases: [declaration(applied, 1)] })).rows[0].declarations + expect(d).toMatchObject({ total: 1, pending: 1, failed: 0, summary: { text: '1 of 1 pending', tone: 'unknown' } }) + } + }) +}) + +it('uses plain WAL evidence without turning an operator success into an archive', () => { + const condition = { type: 'ContinuousArchiving', status: 'True', message: 'Continuous archiving is working', lastTransitionTime: '2026-10-01T12:00:00Z' } + const c = cluster('payments', 'db', { status: { conditions: [condition] } }) + const p = buildCNPGFleet(resp({ clusters: [c] })).rows[0].protection + expect(p.walArchiving).toMatchObject({ text: 'Not archived: no destination configured', operatorCondition: condition }) + expect(p.walArchiving.detail).toBe("WAL is not archived to recovery storage, so point-in-time recovery is unavailable. CloudNativePG still reports archiving as working because, with no destination, it accepts each WAL file without keeping it.") + expect(p.recoveryWindow.text).toBe('None: no backup destination') + c.spec.backup = { barmanObjectStore: { destinationPath: 's3://backups' } } + const configured = buildCNPGFleet(resp({ clusters: [c] })).rows[0].protection.walArchiving + expect(configured).toMatchObject({ tone: 'healthy', source: 'Cluster status · ContinuousArchiving=True' }) + expect(configured.text).toBe('CNPG reports archiving') +}) +it('uses the schedule method blocker in plain recovery summaries, including a mismatched destination', () => { + const c = cluster('payments', 'db') + const schedule = { metadata: { name: 'nightly', namespace: 'db' }, spec: { method: 'barmanObjectStore', cluster: { name: 'payments' }, schedule: '0 0 2 * * *' } } + const data = resp({ clusters: [c], scheduledBackups: [schedule] }, { scheduleReadings: { 'db/nightly': 'every day at 02:00 UTC' } }) + const fact = () => buildCNPGFleet(data).rows[0].protection.schedule + expect(fact()).toMatchObject({ text: 'Enabled · every day at 02:00 UTC · blocked: no backup destination', tone: 'degraded' }) + c.spec.backup = { volumeSnapshot: {} } + expect(fact()).toMatchObject({ text: 'Enabled · every day at 02:00 UTC · blocked: no barmanObjectStore destination', tone: 'degraded' }) + schedule.spec.method = 'volumeSnapshot' + expect(fact()).toMatchObject({ text: 'Enabled · not run yet', tone: 'neutral' }) + delete data.scheduleReadings + expect(fact().text).toBe('Enabled · not run yet') + expect(fact().source).toBe('ScheduledBackup nightly · cron 0 0 2 * * *') + delete (schedule.spec as any).schedule + expect(fact().text).toBe('Enabled · not run yet') + data.objects.backups = [{ apiVersion: 'postgresql.cnpg.io/v1', kind: 'Backup', metadata: { namespace: 'db', name: 'run', labels: { 'cnpg.io/scheduled-backup': 'nightly' } }, spec: { cluster: { name: 'payments' } }, status: { phase: 'completed' } }] + expect(fact().text).toBe('Enabled') +}) + +it('keeps the plain scheduler cause and the complete scheduler message as separate evidence', () => { + const pending = { apiVersion: 'v1', kind: 'Pod', metadata: { name: 'orders-2-join-x', namespace: 'db', labels: { 'cnpg.io/cluster': 'orders', 'cnpg.io/jobRole': 'join', 'cnpg.io/instanceName': 'orders-2' } } } + const raw = '0/2 nodes are available: 2 Too many pods. preemption: no victims.' + const data = { ...resp({ clusters: [cluster('orders', 'db')] }, { issues: [{ id: 'j', severity: 'critical' as const, category: 'unschedulable', kind: 'Pod', namespace: 'db', name: 'orders-2-join-x', reason: 'Unschedulable', message: raw }] }), jobPods: [pending] } + const problem = buildCNPGFleet(data).rows[0].problems.find((p) => p.subject.name === 'orders-2-join-x')! + expect(problem.instance).toBe('orders-2') + expect(problem.detail).toBe('both nodes have reached their Pod limit') + expect(problem.rawDetail).toBe(raw) +}) + +it.each(['False', 'Unknown', undefined])('does not invent operator archiving success for %s', (status) => { + const c = cluster('analytics', 'db', { status: { conditions: status ? [{ type: 'ContinuousArchiving', status }] : [] } }) + expect(buildCNPGFleet(resp({ clusters: [c] })).rows[0].protection.walArchiving.detail).toBe('WAL is not archived to recovery storage, so point-in-time recovery is unavailable.') +}) +it('counts zero ready instance Pods only when the Pod inventory was read', () => { + const c = cluster('analytics', 'db', { spec: { instances: 1 }, status: { readyInstances: undefined } }) + expect(buildCNPGFleet(resp({ clusters: [c], pods: [] })).rows[0].podReadiness).toEqual({ ready: 0, total: 0 }) + expect(buildCNPGFleet(resp({ clusters: [c] }, { coverage: { pods: { state: 'denied' } } })).rows[0].podReadiness).toBeUndefined() +}) + +it('counts an enabled destination-blocked schedule as a Cluster protection problem', () => { + const c = cluster('payments', 'db', { spec: { instances: 1 }, status: { readyInstances: 1, phase: 'Cluster in healthy state' } }) + const schedule = { apiVersion: G, kind: 'ScheduledBackup', metadata: { name: 'payments-nightly', namespace: 'db' }, spec: { cluster: { name: 'payments' } } } + const fleet = buildCNPGFleet(resp({ clusters: [c], scheduledBackups: [schedule] }, { issues: [serverProblem('CNPGScheduleDestinationMissing', 'Backup schedule payments-nightly cannot run: no backup destination', 'ScheduledBackup', 'payments-nightly')] })) + const row = fleet.rows[0] + expect(row.problems).toContainEqual(expect.objectContaining({ title: 'Backup schedule payments-nightly cannot run: no backup destination', severity: 'warning', category: 'protection', subject: expect.objectContaining({ kind: 'ScheduledBackup', name: 'payments-nightly' }) })) + expect(row.attention).toBe(true) + expect(fleet.attentionCount).toBe(1) + expect(fleet.categoryCounts.protection).toBe(1) + expect(row.categories.has('protection')).toBe(true) + expect(cnpgDimensions({ row }).find((d) => d.id === 'protection')?.tone).toBe('degraded') + for (const schedules of [[], [{ ...schedule, spec: { ...schedule.spec, suspend: true } }]]) { + expect(buildCNPGFleet(resp({ clusters: [c], scheduledBackups: schedules })).rows[0].attention).toBe(false) + } + expect(buildCNPGFleet(resp({ clusters: [c], scheduledBackups: [schedule] }, { coverage: { scheduledBackups: { state: 'denied' } } })).attentionCount).toBe(0) +}) + +it('marks a schedule-method mismatch even when the Cluster has another working destination', () => { + const c = cluster('payments', 'db', { spec: { instances: 1, plugins: [{ name: 'barman-cloud.cloudnative-pg.io', isWALArchiver: true, parameters: { barmanObjectName: 'store' } }] }, status: { readyInstances: 1, conditions: [{ type: 'ContinuousArchiving', status: 'True' }] } }) + const schedule = { metadata: { name: 'payments-nightly', namespace: 'db' }, spec: { cluster: { name: 'payments' }, method: 'barmanObjectStore' } } + const row = buildCNPGFleet(resp({ clusters: [c], scheduledBackups: [schedule] }, { issues: [serverProblem('CNPGScheduleDestinationMissing', 'Backup schedule payments-nightly cannot run: no barmanObjectStore destination', 'ScheduledBackup', 'payments-nightly')] })).rows[0] + expect(row.attention).toBe(true) + expect(cnpgDimensions({ row }).find((d) => d.id === 'protection')).toMatchObject({ tone: 'degraded', text: 'Backup schedule payments-nightly cannot run: no barmanObjectStore destination' }) +}) + + +it('uses server findings without classifying cached-object problems again', () => { + const c = cluster('pg-a', 'db', { status: { currentPrimary: 'pg-a-2' } }) + const schedule = { metadata: { name: 'nightly', namespace: 'db' }, spec: { cluster: { name: 'pg-a' } } } + const row = buildCNPGFleet(resp({ clusters: [c], pods: [pod('pg-a-1', 'db', 'pg-a', 'primary', false)], scheduledBackups: [schedule] })).rows[0] + expect(row.readinessContradicted).toBe(true) + expect(row.primaryConflict).toEqual({ status: 'pg-a-2', labelled: 'pg-a-1' }) + expect(row.problems).toEqual([]) +}) + + +it('excludes predecessor backup success, failures and restores from a recreated Cluster', () => { + const c = cluster('pg-a', 'db', { metadata: { uid: 'current', creationTimestamp: '2026-10-01T00:00:00Z' }, spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', isWALArchiver: true, parameters: { barmanObjectName: 'store' } }] } }) + const old = { apiVersion: G, kind: 'Backup', metadata: { name: 'old', namespace: 'db', creationTimestamp: '2026-10-01T01:00:00Z' }, spec: { cluster: { name: 'pg-a' } }, status: { phase: 'completed', startedAt: '2026-10-01T01:00:00Z', stoppedAt: '2026-10-01T01:01:00Z', pluginMetadata: { clusterUID: 'previous' } } } + const failed = { ...old, metadata: { ...old.metadata, name: 'old-failed' }, status: { ...old.status, phase: 'failed' } } + const restored = cluster('restored', 'db', { spec: { bootstrap: { recovery: { backup: { name: 'old' } } } } }) + const store = { apiVersion: 'barmancloud.cnpg.io/v1', kind: 'ObjectStore', metadata: { name: 'store', namespace: 'db' }, status: { serverRecoveryWindow: { 'pg-a': { lastSuccessfulBackupTime: '2026-09-30T23:00:00Z' } } } } + const data = resp({ clusters: [c, restored], backups: [old, failed], objectStores: [store] }, { issues: [serverProblem('CNPGBackupFailed', 'Previous backup failed', 'Backup', 'old-failed')] }) + const row = buildCNPGFleet(data).rows.find((r) => r.name === 'pg-a')! + expect(row.protection.lastSuccessfulBackup.text).toBe('No successful backup yet') + expect(row.protection.restoreValidation.text).toBe('None recorded') + expect(row.problems.some((p) => p.subject.name === 'old-failed')).toBe(false) + const current = { ...old, status: { ...old.status, pluginMetadata: { clusterUID: 'current' } } } + expect(buildCNPGFleet({ ...data, objects: { ...data.objects, backups: [current, failed] } }).rows.find((r) => r.name === 'pg-a')!.protection.lastSuccessfulBackup).toMatchObject({ text: 'Completed', source: 'Backup old' }) +}) + +it('does not attribute predecessor plugin restores or notes to a recreated source', () => { + const source = cluster('pg-a', 'db', { metadata: { uid: 'current', creationTimestamp: '2026-10-01T00:00:00Z' }, spec: { plugins: [{ name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store', serverName: 'archive' } }] } }) + const restored = cluster('restored', 'db', { + metadata: { uid: 'restored', creationTimestamp: '2026-10-02T00:00:00Z' }, + spec: { bootstrap: { recovery: { source: 'origin' } }, externalClusters: [{ name: 'origin', plugin: { name: 'barman-cloud.cloudnative-pg.io', parameters: { barmanObjectName: 'store', serverName: 'archive' } } }] }, + }) + const fact = (target: any, backups: any[] = []) => buildCNPGFleet(resp({ clusters: [source, target], backups })).rows.find((r) => r.name === 'pg-a')!.protection.restoreValidation + const beforeSource = { ...restored, metadata: { ...restored.metadata, creationTimestamp: '2026-09-30T00:00:00Z' } } + expect(cnpgRecoveryMatchesCluster(beforeSource, source, [])).toBe(false) + expect(fact(beforeSource).text).toBe('None recorded') + const pinned = { ...restored, spec: { ...restored.spec, bootstrap: { recovery: { source: 'origin', recoveryTarget: { backupID: 'old-id' } } } } } + const oldBackup = { apiVersion: G, kind: 'Backup', metadata: { name: 'old', namespace: 'db' }, spec: { cluster: { name: 'pg-a' } }, status: { backupId: 'old-id', pluginMetadata: { clusterUID: 'previous' } } } + expect(cnpgRecoveryMatchesCluster(pinned, source, [oldBackup])).toBe(false) + expect(fact(pinned, [oldBackup]).text).toBe('None recorded') + const pitr = { ...restored, spec: { ...restored.spec, bootstrap: { recovery: { source: 'origin', recoveryTarget: { targetTime: '2026-09-30T23:00:00Z' } } } } } + expect(fact(pitr).text).toBe('None recorded') + const withNote = (sourceUID?: string, targetUID = 'restored') => ({ ...restored, metadata: { ...restored.metadata, annotations: { 'radar.skyhook.io/restore-validation': JSON.stringify({ recordedAt: '2026-10-02T02:00:00Z', checked: 'application data', source: { namespace: 'db', name: 'pg-a', uid: sourceUID, verified: !!sourceUID }, target: { namespace: 'db', name: 'restored', uid: targetUID, verified: true } }) } } }) + expect(fact(withNote('previous')).text).toBe('None recorded') + expect(fact(withNote()).text).toBe('Archive restored into restored') + expect(fact(withNote('current')).text).toBe('Validation recorded') + const copied = withNote('current', 'previous-target') + expect(getCNPGRestoreValidation(copied)).toBeNull() + expect(fact(copied).text).toBe('Archive restored into restored') + expect(fact(restored).source).toContain('same archive currently configured') + expect(fact({ ...restored, status: { readyInstances: 0 } }).text).toBe('Archive recovery declared in restored') + const currentBackup = { ...oldBackup, status: { ...oldBackup.status, pluginMetadata: { clusterUID: 'current' } } } + expect(cnpgRecoveryMatchesCluster(pinned, source, [currentBackup])).toBe(true) +}) + +it('matches in-tree recovery by the actual archive and endpoint rather than server name alone', () => { + const archive = { destinationPath: 's3://bucket/prefix/', endpointURL: 'https://s3.example/', serverName: 'pg-a' } + const source = cluster('pg-a', 'db', { spec: { backup: { barmanObjectStore: archive } } }) + const restored = cluster('restored', 'db', { spec: { bootstrap: { recovery: { source: 'origin' } }, externalClusters: [{ name: 'origin', barmanObjectStore: { ...archive, destinationPath: 's3://bucket/prefix' } }] } }) + expect(cnpgRecoveryMatchesCluster(restored, source, [])).toBe(true) + const other = { ...source, spec: { backup: { barmanObjectStore: { ...archive, endpointURL: 'https://other.example' } } } } + expect(cnpgRecoveryMatchesCluster(restored, other, [])).toBe(false) }) diff --git a/packages/k8s-ui/src/components/cnpg/workspace.ts b/packages/k8s-ui/src/components/cnpg/workspace.ts index eb8b1a95c7..4f8ba2fe4f 100644 --- a/packages/k8s-ui/src/components/cnpg/workspace.ts +++ b/packages/k8s-ui/src/components/cnpg/workspace.ts @@ -1,13 +1,23 @@ +import { cnpgBackupDeclaration, cnpgBarmanPlugin } from '../../utils/cnpg-backup' // CloudNativePG workspace model: pure derivations over the /api/cnpg/workspace // payload. Every fact here is something the cluster actually reports; when it // does not report something the value is "unknown", never zero or healthy. -import type { HealthLevel } from '../resources/resource-utils' +import { backupsForScheduledBackup, cnpgScheduleDestinationBlocker, cnpgBackupMatchesCluster, cnpgArchiveMatchesCluster, targetCluster } from './relations' +import { cnpgRoleState } from './databaseRole' +import { formatAge, summarizeSchedulerMessage, type HealthLevel } from '../resources/resource-utils' +import { worseTone } from '../ui/status-tone' +import { type Fact } from '../facts' +import { type ProblemOrigin, type WorkspaceProblem } from '../problems' +import type { ResourceRef } from '../../types/core' +import { formatGrant, type Grant } from '../../utils/grant' +import { issueReasonTitle, issueTitle } from '../issues/severity' import { CNPG_BARMAN_PLUGIN_NAME, getCNPGClusterBackupConfig, getCNPGClusterBarmanPlugin, getCNPGClusterImageTag, + getCNPGClusterIsReplica, getCNPGClusterStatus, getCNPGObjectStoreRecoveryWindows, isApiGroup, @@ -21,6 +31,7 @@ export const CNPG_WORKSPACE_KEYS = [ 'databases', 'publications', 'subscriptions', + 'databaseRoles', 'imageCatalogs', 'clusterImageCatalogs', 'objectStores', @@ -29,12 +40,14 @@ export const CNPG_WORKSPACE_KEYS = [ export type CNPGWorkspaceKey = (typeof CNPG_WORKSPACE_KEYS)[number] -export type CNPGCoverageState = 'full' | 'partial' | 'denied' | 'notInstalled' | 'syncing' | 'error' +export type CNPGCoverageState = 'full' | 'partial' | 'denied' | 'notInstalled' | 'syncing' | 'uncached' | 'error' export interface CNPGKindCoverage { state: CNPGCoverageState /** Denied namespaces, named only when the caller supplied the candidate list. */ deniedNamespaces?: string[] + /** Namespaces the caller may read but Radar's cache does not hold, named under the same rule. */ + uncachedNamespaces?: string[] /** For partial coverage: the namespaces that were read. */ allowedNamespaces?: string[] } @@ -73,6 +86,14 @@ export interface CNPGWorkspaceResponse { issues: CNPGWorkspaceIssue[] audit: CNPGAuditFinding[] backupsOmitted: number + /** Each ScheduledBackup's schedule as the operator reads it, keyed "namespace/name"; absent when it cannot be parsed. */ + scheduleReadings?: Record + /** The manager of each object that has one, keyed "Kind/namespace/name". */ + managedBy?: Record + /** Pods of the Jobs a Cluster controls (initdb, join, restore…), never counted as instances. */ + jobPods?: any[] + /** Where the caller's Jobs were read; a Job Pod is returned only there. Absent from a Radar that predates it. */ + jobCoverage?: CNPGKindCoverage } export const CNPG_KIND_BY_KEY: Record = { @@ -83,6 +104,7 @@ export const CNPG_KIND_BY_KEY: Record k.group !== '' && k.group === (group ?? '') && k.kind === kind) } -/** The value is observed, derived, or not available from the cluster. */ -export type CNPGFactTone = HealthLevel - -export interface CNPGFact { - text: string - tone: CNPGFactTone - /** Where the value comes from, shown next to it so claims carry their source. */ - source?: string - /** A timestamp the text refers to; the UI renders it as an age. */ - at?: string -} - export type CNPGProblemCategory = 'availability' | 'protection' | 'declarations' | 'pooling' export const CNPG_PROBLEM_CATEGORIES: { id: CNPGProblemCategory; label: string }[] = [ { id: 'availability', label: 'Availability' }, - { id: 'protection', label: 'Protection' }, + { id: 'protection', label: 'Backups' }, { id: 'declarations', label: 'Declarations' }, { id: 'pooling', label: 'Pooling' }, ] -export interface CNPGProblem { - /** Stable identity for keys. */ - id: string - severity: 'critical' | 'warning' | 'posture' - category: CNPGProblemCategory - title: string - detail?: string - /** The object the evidence is about (may be the Cluster or a child object). */ - subject: { kind: string; group: string; namespace: string; name: string } - source: 'issue' | 'audit' +/** + * A problem in the CloudNativePG workspace, categorised by the workspace's + * screens. `instance` names the instance a Cluster-level problem is about, + * e.g. the standby an HA slot is kept for. + */ +export type CNPGProblem = WorkspaceProblem & { + reason?: string + slot?: string + instance?: string + /** The cnpg.io/jobRole of the Cluster's own Job whose Pod this is about: a cause, not an instance's symptom. */ + job?: string + /** The Cluster's own Ready condition: a roll-up of the other problems, shown after them. */ + rollup?: boolean } export interface CNPGInstance { @@ -135,15 +148,15 @@ export interface CNPGInstance { } export interface CNPGProtectionFacts { - schedule: CNPGFact & { names: string[] } - destination: CNPGFact & { + schedule: Fact & { names: string[] } + destination: Fact & { method: 'plugin' | 'barmanObjectStore' | 'volumeSnapshot' | 'none' objectStore?: string } - lastSuccessfulBackup: CNPGFact - walArchiving: CNPGFact - recoveryWindow: CNPGFact & { from?: string } - restoreValidation: CNPGFact & { restoredInto?: { namespace: string; name: string } } + lastSuccessfulBackup: Fact + walArchiving: Fact & { state?: 'no_destination' | 'failing' | 'archiving' | 'unknown'; operatorCondition?: { type: string; status: string; message?: string; lastTransitionTime?: string } } + recoveryWindow: Fact & { from?: string } + restoreValidation: Fact & { restoredInto?: { namespace: string; name: string } } } export interface CNPGFleetRow { @@ -153,15 +166,26 @@ export interface CNPGFleetRow { cluster: any controllerStatus: { text: string; level: HealthLevel } instances: { ready: number | null; desired: number | null } + /** + * Ready instances counted from the instance Pods' Ready condition; absent + * when Pods are not readable here or any Pod's readiness is unknown. + */ + podReadiness?: { ready: number; total: number } + /** status.readyInstances claims more ready instances than the Pods show: CNPG status is stale or lagging. */ + readinessContradicted?: boolean + /** status.currentPrimary is not the Pod labelled primary; status may be stale, or a failover is under way. */ + primaryConflict?: { status: string; labelled: string } pods: CNPGInstance[] replicaCluster: { source?: string } | null hibernated: boolean pgVersion: string | null catalog: { kind: string; name: string } | null - replication: CNPGFact - protection: CNPGProtectionFacts & { summary: CNPGFact } - declarations: { summary: CNPGFact; total: number; failed: number; pending: number } + replication: Fact + protection: CNPGProtectionFacts & { summary: Fact } + declarations: { summary: Fact; total: number; failed: number; pending: number } poolers: string[] + /** The Pooler objects behind `poolers`, for their type and Service port. */ + poolerObjects?: any[] /** False when Poolers are not readable in this cluster's namespace, so an empty list means unknown. */ poolersKnown: boolean problems: CNPGProblem[] @@ -169,7 +193,12 @@ export interface CNPGFleetRow { attention: boolean categories: Set /** GitOps owner recorded on the Cluster, when it carries the standard labels. */ - gitops: CNPGGitOpsSource | null + /** The GitOps or Helm object that manages the Cluster, as the server detected it. */ + managedBy?: ResourceRef + /** Fullest measured volume, set by applyCNPGDisk; absent when no disk reading was requested. */ + disk?: Fact + /** Growth of the fastest-growing volume, set by applyCNPGFleetMetrics when measured. */ + diskGrowth?: Fact } export interface CNPGFleet { @@ -185,8 +214,195 @@ const PROTECTION_ISSUE_REASONS = new Set([ 'CNPGLastBackupFailed', 'CNPGBackupFailed', 'CNPGScheduledBackupMissed', + 'CNPGScheduledRunNoBackup', + 'CNPGScheduleDestinationMissing', ]) +// What an instance Pod's bare reason means, said about the Pod. +const CNPG_POD_REASON_SENTENCES: Record = { + ReadinessProbeFailed: 'not ready (readiness probe failing)', + // The issue does not say whether the Pod is serving now (it may have come + // back within the settle window), so restarts are worded as past. + LivenessProbeFailed: 'restarted recently (liveness probe failing)', + CrashLoopBackOff: 'restarted recently (CrashLoopBackOff)', + HighRestartCount: 'restarted repeatedly', + OOMKilled: 'killed for running out of memory (OOMKilled)', + ImagePullBackOff: 'cannot pull its image (ImagePullBackOff)', + ErrImagePull: 'cannot pull its image (ErrImagePull)', +} + +// Plain headlines for CNPG issues whose message carries the operator's own +// condition text; that message becomes the detail beneath. +// Reasons the Issues page already titles come from issueReasonTitle, so a +// problem reads the same here and there; these are the rest. +const CNPG_REASON_TITLES: Record = { + CNPGBackupFailed: 'Backup failed', + CNPGScheduledBackupMissed: 'A scheduled backup did not run', + CNPGCertificateExpiring: 'A certificate expires soon', + CNPGCertificateExpired: 'A certificate has expired', +} + +/** + * A problem's headline and detail from an issue. Known CNPG reasons get a + * short plain title with the operator's message beneath; otherwise the + * message is the title, unless it is empty or only the reason token (e.g. + * "ReadinessProbeFailed"), which is turned into a sentence about the subject. + */ +export function cnpgIssueText(issue: Pick): { title: string; detail?: string } { + let message = issue.message?.trim() ?? '' + const cause = issue.cause?.trim() || undefined + // The run's time is the issue's first_seen, not part of the message. + if (issue.reason === 'CNPGScheduledRunNoBackup' && issue.first_seen && message) message = `${message} ${formatAge(issue.first_seen)} ago` + if (['CNPGScheduleDestinationMissing', 'CNPGInstanceReadinessMismatch', 'CNPGPrimaryLabelMismatch'].includes(issue.reason)) return { title: message, detail: cause } + const known = issueReasonTitle(issue.reason) ?? CNPG_REASON_TITLES[issue.reason] + if (known) return { title: known, detail: [stripTitlePrefix(message, known), cause].filter(Boolean).join(' ') || undefined } + if (message && message !== issue.reason && /\s/.test(message)) return { title: message, detail: cause } + const token = message || issue.reason + const sentence = CNPG_POD_REASON_SENTENCES[token] + if (sentence) return { title: `${issue.name} ${sentence}`, detail: cause } + const words = token.replace(/([a-z])([A-Z])/g, '$1 $2').toLowerCase() + return { title: `${issue.kind} ${issue.name}: ${words}`, detail: cause } +} + +// "Backup failed: cannot proceed…" under the title "Backup failed" repeats it. +function stripTitlePrefix(message: string, title: string): string { + if (!message.toLowerCase().startsWith(title.toLowerCase())) return message + const rest = message.slice(title.length).replace(/^[\s:;,.\-–—]+/, '') + return rest ? rest[0].toUpperCase() + rest.slice(1) : '' +} + +/** + * Failed Backups of one cluster that failed for the same reason become one + * problem: "3 backups failed: ", about the latest of them, the others + * named in alsoAbout. Each Backup is otherwise its own issue, and a schedule + * failing every night would list the same sentence over and over. + */ +type IssueProblem = CNPGProblem + +export function cnpgCollapseBackupFailures(problems: IssueProblem[], backupTimes: Map = new Map()): IssueProblem[] { + const groups = new Map() + const out: IssueProblem[] = [] + for (const p of problems) { + if (p.reason !== 'CNPGBackupFailed' || p.subject.kind !== 'Backup') { + out.push(p) + continue + } + const key = `${p.severity}\x00${p.detail ?? ''}` + groups.set(key, [...(groups.get(key) ?? []), p]) + } + for (const list of groups.values()) { + if (list.length === 1) { + out.push(list[0]) + continue + } + const at = (p: IssueProblem) => backupTimes.get(p.subject.name) ?? 0 + const sorted = [...list].sort((a, b) => at(b) - at(a) || b.subject.name.localeCompare(a.subject.name)) + const latest = sorted[0] + out.push({ + ...latest, + id: `backups-failed:${latest.subject.namespace}:${latest.detail ?? ''}`, + title: latest.detail ? `${list.length} backups failed: ${latest.detail[0].toLowerCase()}${latest.detail.slice(1)}` : `${list.length} backups failed`, + detail: undefined, + alsoAbout: sorted.slice(1).map((p) => ({ kind: p.subject.kind, name: p.subject.name })), + }) + } + return out +} + +/** + * "Latest backup failed" restates a failed-Backup problem when that problem + * is about the cluster's newest Backup, so the duplicate is dropped and the + * count stays honest. With the newest Backup unknown, both stay. + */ +export function cnpgFoldLastBackupFailed(problems: IssueProblem[], newestBackup: string | undefined): CNPGProblem[] { + const covered = + !!newestBackup && + problems.some( + (p) => + p.reason === 'CNPGBackupFailed' && + p.subject.kind === 'Backup' && + (p.subject.name === newestBackup || p.alsoAbout?.some((o) => o.kind === 'Backup' && o.name === newestBackup)), + ) + return problems + .filter((p) => !(covered && p.reason === 'CNPGLastBackupFailed')) +} + +// When each of the cluster's Backups started (status.startedAt, else its +// creation), by name: the issues about Backups carry no time of their own. +function backupTimesOf(cluster: any, backups: any[]): Map { + const out = new Map() + for (const b of backups) { + if (!cnpgBackupMatchesCluster(b, cluster)) continue + out.set(b.metadata.name, Date.parse(b?.status?.startedAt ?? b?.metadata?.creationTimestamp ?? '') || 0) + } + return out +} + + +// Each entry names what the Go detector (internal/issues/source_cnpg*.go and +// the Pod detector) actually reads. "Reported by CNPG" only where the operator +// itself wrote the failure; a threshold or comparison Radar applies is a +// "Radar check". +const CNPG_CONDITION_ORIGINS: Record = { + CNPGLastBackupFailed: 'Cluster LastBackupSucceeded condition', + CNPGClusterTerminal: 'Cluster status.phase', + CNPGClusterUnrecoverable: 'Cluster status.phase', + CNPGClusterPluginFailure: 'Cluster status.phase', + CNPGClusterFailingOver: 'Cluster status.phase', + CNPGClusterWaitingForUser: 'Cluster status.phase', + CNPGDeclarativeNotApplied: 'status.applied and status.message', +} + +const POD_ORIGINS: Record = { + ReadinessProbeFailed: { label: 'Pod readiness probe', detail: 'Kubelet probe-failure events and the Pod\'s Ready condition' }, + LivenessProbeFailed: { label: 'Pod liveness probe', detail: 'Kubelet probe-failure events and container restarts' }, + ReadinessProbeInvalid: { label: 'Radar check of the probe', detail: 'The readiness probe names a port the container does not declare' }, + LivenessProbeInvalid: { label: 'Radar check of the probe', detail: 'The liveness probe names a port the container does not declare' }, + HighRestartCount: { label: 'Radar check of restarts', detail: 'More than 3 restarts on a container that is still unhealthy' }, + InitContainerStalled: { label: 'Radar check of init containers', detail: 'An init container has not finished' }, + Unschedulable: { label: 'Kubernetes scheduler', detail: "The Pod's PodScheduled condition (reason Unschedulable) and the scheduler's message" }, +} + +/** + * Where an issue's evidence comes from, in user terms: what CloudNativePG + * reported, a Backup's or Pod's own status, or Radar's own check. A reason this + * does not know reads "Detected by Radar" rather than a guessed source. + */ +export function cnpgIssueOrigin(issue: Pick): ProblemOrigin { + const condition = CNPG_CONDITION_ORIGINS[issue.reason] + if (condition) return { label: 'Reported by CNPG', detail: condition } + switch (issue.reason) { + case 'CNPGWALArchivingFailing': + return issue.kind === 'Backup' + ? { label: 'Backup status', detail: 'Backup status.phase walArchivingFailing and status.error' } + : { label: 'Reported by CNPG', detail: 'Cluster ContinuousArchiving condition' } + case 'CNPGBackupFailed': + return { label: 'Backup status', detail: 'Backup status.phase and status.error' } + case 'CNPGClusterDegraded': + return { label: 'Radar check of ready instances', detail: 'spec.instances against status.readyInstances, unless the phase, hibernation or fencing explains it' } + case 'CNPGScheduleDestinationMissing': + return { label: 'Radar check of the backup destination', detail: 'ScheduledBackup method against its target Cluster spec' } + case 'CNPGInstanceReadinessMismatch': + return { label: 'Radar check of instance readiness', detail: 'Instance Pod Ready conditions against Cluster status.readyInstances' } + case 'CNPGPrimaryLabelMismatch': + return { label: 'Radar check of the primary', detail: 'Instance Pod role labels against Cluster status.currentPrimary' } + case 'CNPGScheduledRunNoBackup': + return { label: 'Radar check of the backup schedule', detail: 'The schedule, read as the operator does, against the cluster\'s newest successful backup' } + case 'CNPGScheduledBackupMissed': + return { label: 'Radar check of the backup schedule', detail: 'ScheduledBackup status.nextScheduleTime passed more than 10 minutes ago' } + case 'CNPGCertificateExpiring': + case 'CNPGCertificateExpired': + return { label: 'Certificate expiry (from Cluster status)', detail: 'Cluster status.certificates.expirations, compared with now' } + } + if (issue.kind === 'Pod') return POD_ORIGINS[issue.reason] ?? { label: 'Pod status' } + return { label: 'Detected by Radar' } +} + +/** The headline alone; see cnpgIssueText. */ +export function cnpgIssueTitle(issue: Pick): string { + return cnpgIssueText(issue).title +} + export function cnpgIssueCategory(issue: Pick): CNPGProblemCategory { if (PROTECTION_ISSUE_REASONS.has(issue.reason)) return 'protection' switch (issue.kind) { @@ -197,6 +413,7 @@ export function cnpgIssueCategory(issue: Pick | null | undefined, obj: any): ResourceRef | undefined { + const kind = obj?.kind + const name = obj?.metadata?.name + if (!kind || !name) return undefined + return ws?.managedBy?.[`${kind}/${obj?.metadata?.namespace ?? ''}/${name}`] } function scheduleFact( cluster: any, schedules: any[], cov: CNPGKindCoverage, + readings: Record = {}, + backups: any[] = [], ): CNPGProtectionFacts['schedule'] { const ns = cluster.metadata?.namespace if (!coverageReadable(cov, ns)) { - return { text: coverageUnavailableText(cov, 'ScheduledBackups'), tone: 'unknown', names: [] } + return { text: cnpgCoverageGap(cov, 'ScheduledBackups', ns), tone: 'unknown', names: [] } } const mine = schedules.filter((s) => s.metadata?.namespace === ns && specClusterName(s) === cluster.metadata?.name) if (mine.length === 0) return { text: 'No declarative schedule', tone: 'neutral', names: [] } @@ -307,15 +523,21 @@ function scheduleFact( return { text: mine.length === 1 ? 'Schedule suspended' : 'All schedules suspended', tone: 'degraded', names } } const cron = active[0]?.spec?.schedule + const reading = readings[`${ns}/${active[0]?.metadata?.name}`] + const blockers = active.map((s) => cnpgScheduleDestinationBlocker(s, [cluster])).filter((b): b is string => !!b) + const cadence = active.length === 1 ? reading || cron : undefined + const run = active.some((s) => s.status?.lastScheduleTime || backupsForScheduledBackup(s, backups).length > 0) return { - text: active.length === 1 ? (cron ? `Scheduled · ${cron}` : 'Scheduled') : `${active.length} schedules`, - tone: 'healthy', + text: [active.length === 1 ? 'Enabled' : `${active.length} enabled schedules`, blockers.length || run ? cadence : undefined, blockers.length ? `blocked: ${[...new Set(blockers)].map((blocker) => blocker[0].toLowerCase() + blocker.slice(1)).join(', ')}` : !run ? 'not run yet' : undefined].filter(Boolean).join(' · '), + tone: blockers.length ? 'degraded' : 'neutral', names, + ...(active.length === 1 && cron ? { source: `ScheduledBackup ${active[0]?.metadata?.name} · cron ${cron}` } : {}), } } function destinationFact(cluster: any): CNPGProtectionFacts['destination'] { - const plugin = getCNPGClusterBarmanPlugin(cluster) + const declaration = cnpgBackupDeclaration(cluster) + const plugin = cnpgBarmanPlugin(declaration) if (plugin?.barmanObjectName) { return { text: `ObjectStore ${plugin.barmanObjectName}`, @@ -324,11 +546,10 @@ function destinationFact(cluster: any): CNPGProtectionFacts['destination'] { objectStore: plugin.barmanObjectName, } } - const cfg = getCNPGClusterBackupConfig(cluster) - if (cfg.destinationPath) { - return { text: cfg.destinationPath, tone: 'neutral', method: 'barmanObjectStore' } + if (declaration.inTreeDestination) { + return { text: declaration.inTreeDestination, tone: 'neutral', method: 'barmanObjectStore' } } - if (cluster?.spec?.backup?.volumeSnapshot) { + if (declaration.snapshotsConfigured) { return { text: 'Volume snapshots', tone: 'neutral', method: 'volumeSnapshot' } } return { text: 'No destination configured', tone: 'neutral', method: 'none' } @@ -359,14 +580,12 @@ function lastBackupFact( storesUnreadable: CNPGKindCoverage | null, ): CNPGProtectionFacts['lastSuccessfulBackup'] { const ns = cluster.metadata?.namespace - const name = cluster.metadata?.name const candidates: { at: string; source: string }[] = [] if (coverageReadable(backupsCov, ns)) { const completed = backups .filter( (b) => - b.metadata?.namespace === ns && - specClusterName(b) === name && + cnpgBackupMatchesCluster(b, cluster) && isApiGroup(b.apiVersion, 'postgresql.cnpg.io') && b.status?.phase === 'completed', ) @@ -377,26 +596,105 @@ function lastBackupFact( if (window?.lastSuccess) candidates.push({ at: window.lastSuccess, source: `ObjectStore ${window.store} status` }) const cfg = getCNPGClusterBackupConfig(cluster) if (!cfg.plugin && cfg.lastSuccessfulBackup) candidates.push({ at: cfg.lastSuccessfulBackup, source: 'Cluster status' }) - if (candidates.length === 0) { + const createdAt = Date.parse(cluster.metadata?.creationTimestamp ?? '') + const current = candidates.filter((candidate) => !Number.isFinite(createdAt) || Date.parse(candidate.at) >= createdAt) + if (current.length === 0) { if (!coverageReadable(backupsCov, ns)) { - return { text: coverageUnavailableText(backupsCov, 'Backups'), tone: 'unknown' } + return { text: cnpgCoverageGap(backupsCov, 'Backups', ns), tone: 'unknown', source: 'Backups not read' } } - if (storesUnreadable) return { text: coverageUnavailableText(storesUnreadable, 'ObjectStores'), tone: 'unknown' } - return { text: 'None observed', tone: 'unknown' } + if (storesUnreadable) return { text: cnpgCoverageGap(storesUnreadable, 'ObjectStores', ns), tone: 'unknown' } + return { text: 'No successful backup yet', tone: 'degraded', source: 'Backups read in this namespace; none completed' } } - const best = candidates.reduce((a, b) => (Date.parse(a.at) >= Date.parse(b.at) ? a : b)) + const best = current.reduce((a, b) => (Date.parse(a.at) >= Date.parse(b.at) ? a : b)) return { text: 'Completed', tone: 'healthy', at: best.at, source: best.source } } -function walFact(cluster: any): CNPGFact { +export const CNPG_NO_WAL_ARCHIVE_DESTINATION = 'Not archived: no destination configured' + +function walFact(cluster: any): CNPGProtectionFacts['walArchiving'] { const conds = cluster?.status?.conditions const c = Array.isArray(conds) ? conds.find((x: any) => x?.type === 'ContinuousArchiving') : null - if (!c) return { text: 'Not reported', tone: 'unknown', source: 'Cluster status' } - if (c.status === 'True') return { text: 'Archiving', tone: 'healthy', source: 'ContinuousArchiving condition' } - if (c.status === 'False') { - return { text: c.message ? `Failing · ${c.message}` : 'Failing', tone: 'unhealthy', source: 'ContinuousArchiving condition' } + const declaration = cnpgBackupDeclaration(cluster) + const plugin = cnpgBarmanPlugin(declaration) + const archivers = declaration.plugins.filter((p) => p.enabled && p.isWALArchiver) + const archiveConfigured = archivers.length > 0 || !!declaration.inTreeDestination + if (c?.status === 'False' && archiveConfigured) { + return { state: 'failing', text: 'Failing', tone: 'unhealthy', source: 'ContinuousArchiving condition', ...(c.lastTransitionTime ? { at: c.lastTransitionTime, atMeaning: 'since' as const } : {}), ...(c.message ? { detail: c.message } : {}) } } - return { text: 'Unknown', tone: 'unknown', source: 'ContinuousArchiving condition' } + const destinationKnown = !!(plugin?.isWALArchiver && plugin.barmanObjectName) || !!declaration.inTreeDestination + const customArchiver = archivers.some((p: any) => p.name !== plugin?.name) + if (!destinationKnown && !customArchiver) { + return { + state: 'no_destination', text: CNPG_NO_WAL_ARCHIVE_DESTINATION, + tone: 'neutral', + source: 'Cluster spec', + detail: "WAL is not archived to recovery storage, so point-in-time recovery is unavailable." + (c?.status === 'True' ? " CloudNativePG still reports archiving as working because, with no destination, it accepts each WAL file without keeping it." : ''), + ...(c ? { operatorCondition: { type: c.type, status: c.status, message: c.message, lastTransitionTime: c.lastTransitionTime } } : {}), + } + } + if (!c) return { state: 'unknown', text: 'Not reported', tone: 'unknown', source: 'Cluster status' } + if (c.status === 'True') { + return { state: 'archiving', text: 'CNPG reports archiving', tone: 'healthy', source: 'Cluster status · ContinuousArchiving=True', ...(customArchiver && !destinationKnown ? { detail: 'Archive plugin declared; its destination is not assessed here' } : {}), ...(c.lastTransitionTime ? { at: c.lastTransitionTime, atMeaning: 'since' as const } : {}) } + } + return { state: 'unknown', text: 'Unknown', tone: 'unknown', source: 'ContinuousArchiving condition' } +} + +export const CNPG_RESTORE_VALIDATION_ANNOTATION = 'radar.skyhook.io/restore-validation' + +export interface CNPGRestoreValidationNote { + recordedAt: string + recordedBy?: string + checked: string + targetTime?: string + source?: { namespace: string; name: string; uid?: string; verified: boolean } + target?: { namespace: string; name: string; uid?: string; verified: boolean } +} + +/** The validation note recorded on a restored Cluster, or null when absent or malformed. */ +export function getCNPGRestoreValidation(cluster: any): CNPGRestoreValidationNote | null { + const raw = cluster?.metadata?.annotations?.[CNPG_RESTORE_VALIDATION_ANNOTATION] + if (typeof raw !== 'string' || !raw.trim()) return null + try { + const v = JSON.parse(raw) + if (typeof v?.recordedAt !== 'string' || typeof v?.checked !== 'string' || !v.checked) return null + if (v.target?.uid && v.target.uid !== cluster.metadata?.uid) return null + return v as CNPGRestoreValidationNote + } catch { + return null + } +} + +function noteIsAbout(note: CNPGRestoreValidationNote, cluster: any): boolean { + return !!note.source?.uid && note.source.uid === cluster.metadata?.uid +} + +/** A recovery declaration compatible with this live source, excluding known predecessor evidence. */ +export function cnpgRecoveryMatchesCluster(restored: any, cluster: any, backups: any[]): boolean { + if (restored === cluster || restored.metadata?.namespace !== cluster.metadata?.namespace) return false + const recovery = restored.spec?.bootstrap?.recovery + if (!recovery) return false + const created = Date.parse(cluster.metadata?.creationTimestamp ?? '') + const restoredAt = Date.parse(restored.metadata?.creationTimestamp ?? '') + const targetAt = Date.parse(recovery.recoveryTarget?.targetTime ?? '') + if (Number.isFinite(created) && (restoredAt < created || targetAt < created)) return false + const note = getCNPGRestoreValidation(restored) + if (note?.source?.uid && !noteIsAbout(note, cluster)) return false + const ns = cluster.metadata?.namespace + if (recovery.backup?.name) { + const backup = backups.find((b) => b.metadata?.namespace === ns && b.metadata?.name === recovery.backup.name) + return cnpgBackupMatchesCluster(backup, cluster) + } + const source = (restored.spec?.externalClusters ?? []).find((e: any) => e?.name === recovery.source) + if (!source) return false + const params = source.plugin?.name === CNPG_BARMAN_PLUGIN_NAME ? source.plugin.parameters : undefined + const external = source.barmanObjectStore + const archiveMatch = params?.barmanObjectName + ? cnpgArchiveMatchesCluster(cluster, { kind: 'objectStore', objectStore: params.barmanObjectName, serverName: params.serverName || recovery.source }) + : external && cnpgArchiveMatchesCluster(cluster, { kind: 'inTree', barmanObjectStore: external, serverName: external.serverName || recovery.source }) + if (!archiveMatch) return false + const id = recovery.recoveryTarget?.backupID + const pinned = id ? backups.filter((b) => b.metadata?.namespace === ns && b.spec?.cluster?.name === cluster.metadata?.name && b.status?.backupId === id) : [] + return pinned.length === 0 || pinned.some((b) => cnpgBackupMatchesCluster(b, cluster)) } function restoreValidationFact( @@ -405,26 +703,24 @@ function restoreValidationFact( backups: any[], backupsReadable: boolean, ): CNPGProtectionFacts['restoreValidation'] { - const plugin = getCNPGClusterBarmanPlugin(cluster) - const server = plugin?.serverName || cluster.metadata?.name - const store = plugin?.barmanObjectName const ns = cluster.metadata?.namespace - const name = cluster.metadata?.name - const restored = allClusters.find((c) => { - if (c === cluster || c.metadata?.namespace !== ns) return false - const recovery = c.spec?.bootstrap?.recovery - if (!recovery) return false - const sourceName = recovery.source - if (sourceName) { - const ext = (c.spec?.externalClusters ?? []).find((e: any) => e?.name === sourceName) - const params = ext?.plugin?.name === CNPG_BARMAN_PLUGIN_NAME ? ext.plugin.parameters : undefined - if (store && params?.barmanObjectName === store && (params?.serverName || sourceName) === server) return true + const restoredFromThis = allClusters.filter((c) => cnpgRecoveryMatchesCluster(c, cluster, backups)) + const noted = restoredFromThis + .map((c) => ({ c, note: getCNPGRestoreValidation(c) })) + .filter((x): x is { c: any; note: CNPGRestoreValidationNote } => !!x.note && noteIsAbout(x.note, cluster)) + .sort((a, b) => Date.parse(b.note.recordedAt) - Date.parse(a.note.recordedAt))[0] + if (noted) { + const rname = noted.c.metadata?.name + const by = noted.note.recordedBy ? `by ${noted.note.recordedBy}` : 'by a user Radar could not identify' + return { + text: 'Validation recorded', + tone: 'neutral', + at: noted.note.recordedAt, + source: `Recorded ${by} on ${rname}${noted.note.targetTime ? ` (target ${noted.note.targetTime})` : ''}: ${noted.note.checked.length > 140 ? `${noted.note.checked.slice(0, 140)}…` : noted.note.checked}. A person's note, not a check Radar ran.`, + restoredInto: { namespace: noted.c.metadata?.namespace, name: rname }, } - const backupName = recovery.backup?.name - if (!backupName) return false - const backup = backups.find((b) => b.metadata?.namespace === ns && b.metadata?.name === backupName) - return specClusterName(backup) === name - }) + } + const restored = restoredFromThis[0] if (!restored) { // A recovery by Backup name is only attributable when that Backup could be read. const unresolved = !backupsReadable && allClusters.some((c) => c !== cluster && c.metadata?.namespace === ns && c.spec?.bootstrap?.recovery?.backup?.name) @@ -432,24 +728,26 @@ function restoreValidationFact( return { text: 'None recorded', tone: 'unknown', source: 'Kubernetes does not record restore tests' } } const rname = restored.metadata?.name + const fromArchive = !!restored.spec?.bootstrap?.recovery?.source + const sourceDescription = fromArchive ? 'the same archive currently configured for this Cluster' : "this cluster's backups" const ready = typeof restored.status?.readyInstances === 'number' && restored.status.readyInstances > 0 if (!ready) { return { - text: `Recovery declared in ${rname}`, + text: `${fromArchive ? 'Archive recovery' : 'Recovery'} declared in ${rname}`, tone: 'unknown', - source: `Cluster ${rname} bootstraps from this cluster's backups but has no ready instance yet`, + source: `Cluster ${rname} bootstraps from ${sourceDescription} but has no ready instance yet`, restoredInto: { namespace: restored.metadata?.namespace, name: rname }, } } return { - text: `Restored into ${rname}`, + text: `${fromArchive ? 'Archive restored' : 'Restored'} into ${rname}`, tone: 'neutral', - source: `Cluster ${rname} bootstrapped from this cluster's backups and has ready instances · created ${restored.metadata?.creationTimestamp ?? 'unknown'}. This proves one recovery, not that today's backups restore.`, + source: `Cluster ${rname} bootstrapped from ${sourceDescription} and has ready instances · created ${restored.metadata?.creationTimestamp ?? 'unknown'}. No validation note is recorded on it; this proves one recovery, not that today's backups restore.`, restoredInto: { namespace: restored.metadata?.namespace, name: rname }, } } -function protectionSummary(p: CNPGProtectionFacts): CNPGFact { +function protectionSummary(p: CNPGProtectionFacts): Fact { if (p.walArchiving.tone === 'unhealthy') return { text: 'WAL archiving failing', tone: 'unhealthy' } if (p.destination.method === 'none' && p.schedule.names.length === 0 && p.schedule.tone !== 'unknown') { return { text: 'No backup destination or schedule', tone: 'neutral' } @@ -475,21 +773,52 @@ function pgVersion(cluster: any): string | null { return typeof major === 'number' ? String(major) : null } -function replicationFact(cluster: any, pods: CNPGInstance[], hibernated: boolean, podsCov: CNPGKindCoverage): CNPGFact { +function replicationFact(cluster: any, pods: CNPGInstance[], hibernated: boolean, podsCov: CNPGKindCoverage): Fact { if (hibernated) return { text: 'Hibernated', tone: 'neutral' } const desired = cluster?.spec?.instances if (desired === 1) return { text: 'Single instance', tone: 'neutral' } if (!coverageReadable(podsCov, cluster?.metadata?.namespace)) { - return { text: coverageUnavailableText(podsCov, 'Pods'), tone: 'unknown' } + return { text: cnpgCoverageGap(podsCov, 'Pods', cluster?.metadata?.namespace), tone: 'unknown' } } const replicas = pods.filter((p) => p.role === 'replica') const readyReplicas = replicas.filter((p) => p.ready === true).length if (replicas.length === 0) return { text: 'No replica pods observed', tone: 'unknown' } return { - text: `${readyReplicas}/${replicas.length} replicas ready · lag unknown`, + text: `${readyReplicas}/${replicas.length} Pods ready · lag unknown`, tone: 'unknown', - source: 'Pod readiness does not show whether a replica is streaming', + source: CNPG_LAG_UNMEASURED_SOURCE, + } +} + +const CNPG_JOB_PURPOSE: Record = { + initdb: 'First instance', + join: 'New standby', + 'full-recovery': 'Restore into', + 'snapshot-recovery': 'Restore into', + pgbasebackup: 'Clone into', + import: 'Import into', + 'major-upgrade': 'Major upgrade of', +} + +/** What a Cluster's own Job is for, naming the instance it builds ("New standby pg-2"). */ +export function cnpgJobPurpose(role: string | undefined, instance: string | undefined, pod: string): string { + const purpose = role ? CNPG_JOB_PURPOSE[role] : undefined + if (purpose && instance) return `${purpose} ${instance}` + return `${role ? `${role} ` : ''}Job Pod ${pod}` +} + +interface CNPGJobPod { + role?: string + instance?: string +} + +function jobPodIndex(resp: CNPGWorkspaceResponse): Map { + const idx = new Map() + for (const p of resp.jobPods ?? []) { + const labels = p?.metadata?.labels ?? {} + idx.set(`${p.metadata?.namespace}/${p.metadata?.name}`, { role: labels['cnpg.io/jobRole'], instance: labels['cnpg.io/instanceName'] }) } + return idx } function problemsFor( @@ -497,25 +826,42 @@ function problemsFor( issues: CNPGWorkspaceIssue[], audit: CNPGAuditFinding[], children: Map, + backupTimes: Map = new Map(), + jobPods: Map = new Map(), ): CNPGProblem[] { const ns = cluster.metadata?.namespace const name = cluster.metadata?.name const out: CNPGProblem[] = [] + const fromIssues: IssueProblem[] = [] for (const issue of issues) { if ((issue.namespace ?? '') !== ns) continue const isSelf = issue.kind === 'Cluster' && issue.name === name const owner = children.get(`${issue.kind}/${ns}/${issue.name}`) if (!isSelf && owner !== name) continue - out.push({ + const job = issue.kind === 'Pod' ? jobPods.get(`${ns}/${issue.name}`) : undefined + const text = cnpgIssueText(issue) + fromIssues.push({ id: `${issue.id}:${issue.kind}/${issue.name}`, - severity: issue.severity, + // A standby that cannot join costs redundancy, not service: the primary keeps serving. + severity: job?.role === 'join' ? 'warning' : issue.severity, category: cnpgIssueCategory(issue), - title: issue.message || issue.reason, - detail: issue.cause || undefined, + ...(job + ? { + title: `${cnpgJobPurpose(job.role, job.instance, issue.name)}: ${issue.category ? issueTitle({ category: issue.category, reason: issue.reason }) : text.title}`, + detail: [issue.message?.trim(), issue.cause?.trim()].filter(Boolean).join(' ') || undefined, + job: job.role ?? 'job', + instance: job.instance, + } + : text), subject: { kind: issue.kind, group: issue.group ?? '', namespace: ns, name: issue.name }, source: 'issue', + origin: cnpgIssueOrigin(issue), + reason: issue.reason, + ...(isSelf && issue.reason.startsWith('Ready:') ? { rollup: true } : {}), }) } + const newestBackup = [...backupTimes.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] + out.push(...cnpgFoldLastBackupFailed(cnpgCollapseBackupFailures(fromIssues, backupTimes), newestBackup)) for (const f of audit) { if (f.kind !== 'Cluster' || f.name !== name || (f.namespace ?? '') !== ns) continue out.push({ @@ -528,8 +874,62 @@ function problemsFor( source: 'audit', }) } - const rank = { critical: 0, warning: 1, posture: 2 } as const - return out.sort((a, b) => rank[a.severity] - rank[b.severity] || a.title.localeCompare(b.title)) + return out.sort(cnpgCompareProblems) +} + +/** + * The ready count to show for a cluster: CNPG's status count, or the Pods' + * own count when the status claims more than the Pods show. + */ +export function cnpgReadyInstances(row: Pick): { text: string; podText?: string; tone?: HealthLevel; note?: string } { + if (row.instances.ready === null) { + return { + text: 'Not reported by the operator', + podText: row.podReadiness + ? row.instances.desired !== null + ? `${row.podReadiness.ready} of ${row.instances.desired} instance Pods ready` + : `${row.podReadiness.ready} instance Pod${row.podReadiness.ready === 1 ? '' : 's'} ready` + : undefined, + } + } + const desired = row.instances.desired ?? '–' + if (row.readinessContradicted && row.podReadiness) { + return { + text: `${row.podReadiness.ready}/${desired}`, + tone: row.podReadiness.ready === 0 ? 'unhealthy' : 'degraded', + note: `Counted from the instance Pods' Ready condition. CNPG status reports ${row.instances.ready} ready, so the status may be stale.`, + } + } + return { text: `${row.instances.ready ?? '–'}/${desired}` } +} + +const CNPG_SEVERITY_RANK = { critical: 0, warning: 1, posture: 2 } as const + +/** + * The order problems are shown in, everywhere: most severe first, then a + * cause before its symptoms. One instance Pod's state (a failing probe, a + * crash loop) is usually the symptom of a cluster-level problem (archiving, + * backups, replication, reconciliation, declarations) or of a Job the Cluster + * runs (an initdb or join Pod that cannot start), so at equal severity those + * come first; the Cluster's own Ready condition sums them all up and comes last. + */ +export function cnpgCompareProblems(a: CNPGProblem, b: CNPGProblem): number { + const symptom = (p: CNPGProblem) => (p.rollup ? 2 : p.subject.kind === 'Pod' && p.subject.group === '' && !p.job ? 1 : 0) + return CNPG_SEVERITY_RANK[a.severity] - CNPG_SEVERITY_RANK[b.severity] || symptom(a) - symptom(b) || a.title.localeCompare(b.title) +} + +function sortProblems(list: CNPGProblem[]): CNPGProblem[] { + return list.sort(cnpgCompareProblems) +} + +function primaryConflictOf(cluster: any, pods: any[]): CNPGFleetRow['primaryConflict'] { + const status = cluster?.status?.currentPrimary + if (!status) return undefined + const labelled = pods + .filter((p) => (p?.metadata?.labels?.['cnpg.io/instanceRole'] ?? p?.metadata?.labels?.role) === 'primary') + .map((p) => p.metadata?.name as string) + if (labelled.length === 0 || labelled.includes(status)) return undefined + return { status, labelled: labelled.sort()[0] } } /** Index "Kind/ns/name" → owning cluster name, from each child's spec.cluster.name. */ @@ -537,6 +937,7 @@ function childIndex(resp: CNPGWorkspaceResponse): Map { const idx = new Map() const add = (kind: string, list: any[] | undefined) => { for (const o of list ?? []) { + if (kind === 'Backup' && !targetCluster(o, resp.objects.clusters ?? [])) continue const c = specClusterName(o) if (c) idx.set(`${kind}/${o.metadata?.namespace}/${o.metadata?.name}`, c) } @@ -547,7 +948,8 @@ function childIndex(resp: CNPGWorkspaceResponse): Map { add('Database', resp.objects.databases) add('Publication', resp.objects.publications) add('Subscription', resp.objects.subscriptions) - for (const p of resp.objects.pods ?? []) { + add('DatabaseRole', resp.objects.databaseRoles) + for (const p of [...(resp.objects.pods ?? []), ...(resp.jobPods ?? [])]) { const c = p?.metadata?.labels?.['cnpg.io/cluster'] if (c) idx.set(`Pod/${p.metadata?.namespace}/${p.metadata?.name}`, c) } @@ -561,21 +963,24 @@ function declarationsFor(cluster: any, resp: CNPGWorkspaceResponse): CNPGFleetRo ['databases', resp.objects.databases ?? []], ['publications', resp.objects.publications ?? []], ['subscriptions', resp.objects.subscriptions ?? []], + ['databaseRoles', resp.objects.databaseRoles ?? []], ] let total = 0 let failed = 0 let pending = 0 - let unreadable = false + const unread: string[] = [] + const labels = { databases: 'Databases', publications: 'Publications', subscriptions: 'Subscriptions', databaseRoles: 'DatabaseRoles' } for (const [k, list] of lists) { if (!coverageReadable(coverageOf(resp, k), ns)) { - if (coverageOf(resp, k).state !== 'notInstalled') unreadable = true + if (coverageOf(resp, k).state !== 'notInstalled') unread.push(labels[k as keyof typeof labels]) continue } for (const o of list) { if (o.metadata?.namespace !== ns || specClusterName(o) !== name) continue total++ - if (o.status?.applied === false) failed++ - else if (o.status?.applied !== true) pending++ + const state = cnpgRoleState(o) + if (state === 'failed') failed++ + else if (state === 'pending') pending++ } } const roleStatus = cluster?.status?.managedRolesStatus @@ -588,9 +993,10 @@ function declarationsFor(cluster: any, resp: CNPGWorkspaceResponse): CNPGFleetRo if (failedRoles.has(r.name)) failed++ else if (!reconciledRoles.has(r.name)) pending++ } - let summary: CNPGFact + const unreadable = unread.length > 0 + let summary: Fact if (total === 0) { - summary = unreadable ? { text: 'No access to some declarations', tone: 'unknown' } : { text: 'None declared', tone: 'neutral' } + summary = { text: 'None declared', tone: 'neutral' } } else if (failed > 0) { summary = { text: `${failed} of ${total} not reconciled`, tone: 'degraded' } } else if (pending > 0) { @@ -598,7 +1004,10 @@ function declarationsFor(cluster: any, resp: CNPGWorkspaceResponse): CNPGFleetRo } else { summary = { text: `${total} reconciled`, tone: 'healthy' } } - if (unreadable && total > 0) summary = { ...summary, source: 'Some declaration kinds are not readable' } + if (unreadable) { + const count = total === 0 ? 'Reconciliation unknown' : failed > 0 ? `≥${failed} not reconciled` : pending > 0 ? `≥${pending} pending` : `≥${total} reconciled` + summary = { text: `${count}; ${unread.join(', ')} not read`, tone: failed > 0 ? 'degraded' : 'unknown' } + } return { summary, total, failed, pending } } @@ -607,6 +1016,7 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { const pods = resp.objects.pods ?? [] const stores = resp.objects.objectStores ?? [] const children = childIndex(resp) + const jobs = jobPodIndex(resp) const poolers = resp.objects.poolers ?? [] const rows: CNPGFleetRow[] = clusters.map((cluster) => { @@ -631,7 +1041,7 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { const storesCov = coverageOf(resp, 'objectStores') const storesUnreadable = !!getCNPGClusterBarmanPlugin(cluster)?.barmanObjectName && !coverageReadable(storesCov, ns) const protection: CNPGProtectionFacts = { - schedule: scheduleFact(cluster, resp.objects.scheduledBackups ?? [], coverageOf(resp, 'scheduledBackups')), + schedule: scheduleFact(cluster, resp.objects.scheduledBackups ?? [], coverageOf(resp, 'scheduledBackups'), resp.scheduleReadings, resp.objects.backups ?? []), destination: destinationFact(cluster), lastSuccessfulBackup: lastBackupFact(cluster, resp.objects.backups ?? [], coverageOf(resp, 'backups'), window, storesUnreadable ? storesCov : null), walArchiving: wal, @@ -645,15 +1055,26 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { source: `ObjectStore ${window.store} status (earliest point)`, } : storesUnreadable - ? { text: coverageUnavailableText(storesCov, 'ObjectStores'), tone: 'unknown' } - : { text: 'Not reported', tone: 'unknown' }, + ? { text: cnpgCoverageGap(storesCov, 'ObjectStores', ns), tone: 'unknown' } + : destinationFact(cluster).method === 'none' ? { text: 'None: no backup destination', tone: 'neutral' } : { text: 'Not reported', tone: 'unknown' }, restoreValidation: restoreValidationFact(cluster, clusters, resp.objects.backups ?? [], coverageReadable(coverageOf(resp, 'backups'), ns)), } - const problems = problemsFor(cluster, resp.issues ?? [], resp.audit ?? [], children) + const podsReadable = coverageReadable(coverageOf(resp, 'pods'), ns) + const podReadiness = + podsReadable && resp.objects.pods !== undefined && instancePods.every((p) => p.ready !== null) + ? { ready: instancePods.filter((p) => p.ready).length, total: instancePods.length } + : undefined + const readinessContradicted = !hibernated && !!podReadiness && readyInstances !== null && podReadiness.ready < readyInstances + const primaryConflict = podsReadable ? primaryConflictOf(cluster, pods.filter((p) => p.metadata?.namespace === ns && p.metadata?.labels?.['cnpg.io/cluster'] === name)) : undefined + let problems = sortProblems([ + ...problemsFor(cluster, resp.issues ?? [], resp.audit ?? [], children, backupTimesOf(cluster, resp.objects.backups ?? []), jobs), + + ]) + problems = problems.map((p) => p.origin?.label === 'Kubernetes scheduler' ? { ...p, detail: summarizeSchedulerMessage(p.detail, { plain: true }), rawDetail: p.detail } : p) const categories = new Set( problems.filter((p) => p.severity !== 'posture').map((p) => p.category), ) - const replica = cluster?.spec?.replica?.enabled ? { source: cluster.spec.replica.source } : null + const replica = cluster && getCNPGClusterIsReplica(cluster) ? { source: cluster.spec.replica.source } : null return { key: key(ns, name), @@ -662,6 +1083,9 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { cluster, controllerStatus: { text: status.text, level: status.level }, instances: { ready: readyInstances, desired }, + ...(podReadiness ? { podReadiness } : {}), + ...(readinessContradicted ? { readinessContradicted } : {}), + ...(primaryConflict ? { primaryConflict } : {}), pods: instancePods, replicaCluster: replica, hibernated, @@ -670,6 +1094,7 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { replication: replicationFact(cluster, instancePods, hibernated, coverageOf(resp, 'pods')), protection: { ...protection, summary: protectionSummary(protection) }, declarations: declarationsFor(cluster, resp), + poolerObjects: poolers.filter((p) => p.metadata?.namespace === ns && specClusterName(p) === name), poolers: poolers .filter((p) => p.metadata?.namespace === ns && specClusterName(p) === name) .map((p) => p.metadata?.name), @@ -677,20 +1102,49 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { problems, attention: problems.some((p) => p.severity !== 'posture'), categories, - gitops: cnpgGitOpsSource(cluster), + managedBy: cnpgManagedBy(resp, cluster), } }) - rows.sort((a, b) => Number(b.attention) - Number(a.attention) || a.namespace.localeCompare(b.namespace) || a.name.localeCompare(b.name)) - - const categoryCounts = { availability: 0, protection: 0, declarations: 0, pooling: 0 } as Record - for (const r of rows) for (const c of r.categories) categoryCounts[c]++ - const incompleteKinds = CNPG_WORKSPACE_KEYS.filter((k) => { const s = coverageOf(resp, k).state - return s === 'partial' || s === 'denied' || s === 'syncing' || s === 'error' + return s === 'partial' || s === 'denied' || s === 'syncing' || s === 'error' || s === 'uncached' }) + return finishFleet(rows, incompleteKinds) +} + +function urgencyOf(row: CNPGFleetRow): { worst: number; urgent: number; total: number } { + let worst = 3 + let urgent = 0 + for (const p of row.problems) { + worst = Math.min(worst, CNPG_SEVERITY_RANK[p.severity]) + if (p.severity !== 'posture') urgent++ + } + return { worst, urgent, total: row.problems.length } +} + +/** + * Worst problem first (critical, warning, posture, none), then the most + * attention-level problems, then all problems, then namespace/name — the same + * order in every filter, so a row never jumps when the filter changes. + */ +export function compareCNPGFleetUrgency(a: CNPGFleetRow, b: CNPGFleetRow): number { + const ua = urgencyOf(a) + const ub = urgencyOf(b) + return ( + ua.worst - ub.worst || + ub.urgent - ua.urgent || + ub.total - ua.total || + a.namespace.localeCompare(b.namespace) || + a.name.localeCompare(b.name) + ) +} + +function finishFleet(rows: CNPGFleetRow[], incompleteKinds: CNPGWorkspaceKey[]): CNPGFleet { + rows.sort(compareCNPGFleetUrgency) + const categoryCounts = { availability: 0, protection: 0, declarations: 0, pooling: 0 } as Record + for (const r of rows) for (const c of r.categories) categoryCounts[c]++ return { rows, attentionCount: rows.filter((r) => r.attention).length, @@ -698,3 +1152,533 @@ export function buildCNPGFleet(resp: CNPGWorkspaceResponse): CNPGFleet { incompleteKinds, } } + +/** One cluster's answer from /api/cnpg/disk. */ +export interface CNPGDiskReading { + namespace: string + name: string + /** ok | partial | noSeries | noPrometheus | denied | unavailable | error | notRead | ambiguous | scopeMismatch */ + state: string + grant?: Grant + reason?: string + claims: number + measured: number + max?: { + claim: string + instance: string + role: string + tablespace?: string + usedBytes: number + capacityBytes: number + ratio: number + } + isolation?: CNPGMetricIsolation +} + +export const CNPG_DISK_WARNING_RATIO = 0.8 +export const CNPG_DISK_CRITICAL_RATIO = 0.9 + +export const CNPG_DISK_SOURCE = 'kubelet volume stats via Prometheus' + +export function cnpgVolumeRoleLabel(role: string, tablespace?: string): string { + switch (role) { + case 'PG_DATA': + return 'data volume' + case 'PG_WAL': + return 'WAL volume' + case 'PG_TABLESPACE': + return tablespace ? `tablespace ${tablespace} volume` : 'tablespace volume' + default: + return 'volume' + } +} + +export function cnpgDiskTone(ratio: number): HealthLevel { + if (ratio >= CNPG_DISK_CRITICAL_RATIO) return 'unhealthy' + if (ratio >= CNPG_DISK_WARNING_RATIO) return 'degraded' + return 'healthy' +} + +/** The fleet and summary "Storage" fact: the fullest measured volume, or why there is none. */ +export function cnpgDiskFact(r: CNPGDiskReading | undefined): Fact { + if (!r) return { text: 'Not read', tone: 'unknown' } + if (r.max && (r.state === 'ok' || r.state === 'partial')) { + const partial = r.state === 'partial' ? ` · ${r.measured} of ${r.claims} volumes measured` : '' + return { + text: `${Math.round(r.max.ratio * 100)}% used`, + tone: cnpgDiskTone(r.max.ratio), + source: `Fullest: ${cnpgVolumeRoleLabel(r.max.role, r.max.tablespace)} of ${r.max.instance}, ${formatBytes(r.max.usedBytes)} of ${formatBytes(r.max.capacityBytes)} · ${CNPG_DISK_SOURCE}${partial}.${isolationCaveat(r.isolation)}`, + } + } + switch (r.state) { + case 'denied': + return { text: 'No access', tone: 'unknown', source: r.grant ? `Needs ${formatGrant(r.grant)}` : r.reason } + case 'noPrometheus': + return { text: 'No usage metrics', tone: 'unknown', source: CNPG_PROMETHEUS_NOT_CONNECTED, detail: r.reason } + case 'noSeries': + case 'ok': + case 'partial': + return { text: 'No usage metrics', tone: 'unknown', source: r.reason ?? 'Used space needs Prometheus with kubelet volume stats' } + case 'notRead': + return { text: 'Not measured', tone: 'unknown', source: r.reason } + default: + return { text: 'Unavailable', tone: 'unknown', source: r.reason } + } +} + +/** + * Joins /api/cnpg/disk into the fleet: every row gets its disk fact, and a + * volume at or past the warning threshold becomes a problem, so the cluster + * needs attention. Only a measurement raises one, and the endpoint returns + * measurements only to callers holding the claim and metrics grants. + */ +export function applyCNPGDisk(fleet: CNPGFleet, readings: CNPGDiskReading[] | undefined): CNPGFleet { + if (!readings) return fleet + const byKey = new Map(readings.map((r) => [key(r.namespace, r.name), r])) + const rows = fleet.rows.map((row) => { + const reading = byKey.get(row.key) + const next: CNPGFleetRow = { ...row, disk: cnpgDiskFact(reading) } + const max = reading?.max + if (!max || max.ratio < CNPG_DISK_WARNING_RATIO || (reading.state !== 'ok' && reading.state !== 'partial')) return next + const problem: CNPGProblem = { + id: `disk:${row.key}`, + severity: max.ratio >= CNPG_DISK_CRITICAL_RATIO ? 'critical' : 'warning', + category: 'availability', + title: `The ${cnpgVolumeRoleLabel(max.role, max.tablespace)} of ${max.instance} is ${Math.round(max.ratio * 100)}% full`, + detail: `${formatBytes(max.usedBytes)} of ${formatBytes(max.capacityBytes)} used, from ${CNPG_DISK_SOURCE}.${isolationCaveat(reading.isolation)}`, + subject: { kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: row.namespace, name: row.name }, + source: 'measurement', + ...(reading.isolation?.mode === 'unverified' ? { measuredBy: 'kubelet, matched by claim name', unverifiedMatch: true } : {}), + } + const problems = [...row.problems, problem].sort(cnpgCompareProblems) + return { ...next, problems, attention: true, categories: new Set([...row.categories, problem.category]) } + }) + return finishFleet(rows, fleet.incompleteKinds) +} + +const CNPG_PLUGIN_PHASES: Record = { + 'Cluster cannot proceed to reconciliation due to an unknown plugin being required': 'unknownPlugin', + 'Cluster cannot proceed to reconciliation due to an error while interacting with plugins': 'pluginError', +} + +/** Whether the Cluster's phase says the operator is stuck on a CNPG-I plugin, and which way. */ +export function cnpgPluginPhase(cluster: any): 'unknownPlugin' | 'pluginError' | null { + return CNPG_PLUGIN_PHASES[cluster?.status?.phase] ?? null +} + +/** The CNPG-I plugins a Cluster names in spec.plugins. */ +export function cnpgClusterPlugins(cluster: any): string[] { + const plugins = cluster?.spec?.plugins + return Array.isArray(plugins) ? plugins.map((p: any) => p?.name).filter((n: unknown): n is string => typeof n === 'string' && !!n) : [] +} + +/** Bytes in binary units with IEC labels (GiB), as every CloudNativePG view prints them. */ +export function cnpgFormatBytes(n: number): string { + const u = ['B', 'KiB', 'MiB', 'GiB', 'TiB'] + let v = n + let i = 0 + while (v >= 1024 && i < u.length - 1) { + v /= 1024 + i++ + } + return `${v.toFixed(v >= 10 || i === 0 ? 0 : 1)} ${u[i]}` +} + +const formatBytes = cnpgFormatBytes + +// --------------------------------------------------------------------------- +// Standbys and replication slots: one finding, whichever source saw it +// --------------------------------------------------------------------------- + +/** + * Retained WAL at which an inactive physical slot becomes a concern. An HA + * slot only goes inactive when its standby stops consuming, and the WAL it + * pins keeps growing until that standby catches up or the slot is dropped. + */ +export const CNPG_SLOT_RETENTION_WARNING_BYTES = 1024 ** 3 + +/** The id of a standby's "not receiving WAL" problem: the fleet (Prometheus) and the cluster page (instance manager) raise the same one. */ +export function cnpgStandbyProblemId(rowKey: string, pod: string): string { + return `standby:${rowKey}:${pod}` +} + +/** The id of an inactive slot's retention problem, shared like cnpgStandbyProblemId. */ +export function cnpgSlotProblemId(rowKey: string, slot: string): string { + return `slot:${rowKey}:${slot}` +} + +function haSlotPrefix(cluster: any): string { + const p = cluster?.spec?.replicationSlots?.highAvailability?.slotPrefix + return typeof p === 'string' && p ? p : '_cnpg_' +} + +/** The HA slot CloudNativePG keeps for an instance: the prefix (default `_cnpg_`) then the instance name with `-` as `_`. */ +export function cnpgHASlotName(cluster: any, instance: string): string { + return `${haSlotPrefix(cluster)}${instance.replace(/-/g, '_')}` +} + +/** The instance a managed HA slot serves, or undefined when the slot is not one of this cluster's HA slots. */ +export function cnpgHASlotInstance(cluster: any, slot: string, instances: string[]): string | undefined { + return instances.find((i) => cnpgHASlotName(cluster, i) === slot) +} + +const CLUSTER_GROUP = 'postgresql.cnpg.io' + +function clusterSubject(row: Pick) { + return { kind: 'Cluster', group: CLUSTER_GROUP, namespace: row.namespace, name: row.name } +} + +export interface CNPGStandbyGap { + pod: string + /** What was observed, in words, e.g. "its WAL receiver is down", "replay is paused". */ + evidence: string[] + measuredBy: string + sourceDetail: string + /** Every expected standby is out: there is no failover target with current data. */ + noneReceiving?: boolean + unverifiedMatch?: boolean +} + +/** A standby that receives nothing from the primary. Its replay lag reads 0 because nothing new arrives to replay. */ +export function cnpgStandbyNotReceivingProblem(row: Pick, gap: CNPGStandbyGap): CNPGProblem { + return { + id: cnpgStandbyProblemId(row.key, gap.pod), + reason: 'CNPGStandbyNotReceiving', + severity: gap.noneReceiving ? 'critical' : 'warning', + category: 'availability', + title: `${gap.pod} is not receiving WAL from the primary`, + shortTitle: `${gap.pod} not receiving WAL`, + detail: `${capitalize(gap.evidence.join('; '))}. While it does not stream, the primary keeps WAL for its slot, and it falls behind unless it replays WAL from the archive; a replay lag of 0 does not mean it is caught up, only that nothing new reached it.`, + subject: { kind: 'Pod', group: '', namespace: row.namespace, name: gap.pod }, + source: 'measurement', + measuredBy: gap.measuredBy, + sourceDetail: gap.sourceDetail, + ...(gap.unverifiedMatch ? { unverifiedMatch: true } : {}), + } +} + +export interface CNPGSlotRetention { + slot: string + /** The instance holding the slot (the primary for an HA slot). */ + pod: string + bytes: number + /** The standby an HA slot serves, when the name matches one of the cluster's instances. */ + standby?: string + measuredBy: string + sourceDetail: string + unverifiedMatch?: boolean +} + +export function cnpgSlotRetentionProblem(row: Pick, r: CNPGSlotRetention): CNPGProblem { + const forWhom = r.standby ? ` for ${r.standby}` : '' + return { + id: cnpgSlotProblemId(row.key, r.slot), + reason: 'CNPGInactiveSlot', + slot: r.slot, + severity: 'warning', + category: 'availability', + title: `Inactive slot ${r.slot} holds ${formatBytes(r.bytes)} of WAL on ${r.pod}${forWhom}`, + shortTitle: `${formatBytes(r.bytes)} of WAL held${forWhom || ` by ${r.slot}`}`, + detail: `PostgreSQL keeps every WAL file the slot still needs until ${r.standby ? `${r.standby} catches up` : 'its consumer catches up'} or the slot is dropped, so this grows while the slot stays inactive. Storage shows it beside the volume's other WAL.`, + subject: clusterSubject(row), + source: 'measurement', + measuredBy: r.measuredBy, + sourceDetail: r.sourceDetail, + ...(r.standby ? { instance: r.standby } : {}), + ...(r.unverifiedMatch ? { unverifiedMatch: true } : {}), + } +} + +/** + * Adds problems to a row, replacing any with the same id, and drops those a + * newer, complete read has disproven (`drop`); attention and categories + * follow from what remains. + */ +export function cnpgWithProblems(row: CNPGFleetRow, added: CNPGProblem[], drop?: (p: CNPGProblem) => boolean): CNPGFleetRow { + if (added.length === 0 && !drop) return row + const ids = new Set(added.map((p) => p.id)) + const problems = [...row.problems.filter((p) => !ids.has(p.id) && !drop?.(p)), ...added].sort(cnpgCompareProblems) + if (problems.length === row.problems.length && problems.every((p, i) => p === row.problems[i])) return row + return { + ...row, + problems, + attention: problems.some((p) => p.severity !== 'posture'), + categories: new Set(problems.filter((p) => p.severity !== 'posture').map((p) => p.category)), + } +} + +function capitalize(s: string): string { + return s ? s[0].toUpperCase() + s.slice(1) : s +} + +/** The short per-fact source when Radar has no Prometheus; the reason goes in `detail`. */ +export const CNPG_PROMETHEUS_NOT_CONNECTED = 'Prometheus not connected' + +const CNPG_LAG_UNMEASURED_SOURCE = 'Pod readiness does not show whether a replica is streaming' + +/** One cluster's answer from /api/cnpg/fleet-metrics. */ +export interface CNPGFleetMetricsReading { + namespace: string + name: string + /** ok | noStandby | noSeries | denied | ambiguous | scopeMismatch | error | notRead */ + lag: { + state: string + grant?: Grant + reason?: string + seconds?: number + pod?: string + /** Standbys whose replay lag was read; `seconds` covers only these. */ + lagStandbys?: number + /** The worst standby's lowest recorded lag over `sustainedWindow`, across every scrape of it; it was already reporting by the window's start. */ + sustainedSeconds?: number + sustainedPod?: string + sustainedWindow?: string + isolation?: CNPGMetricIsolation + /** Instances reporting that they are in recovery (standbys). */ + standbys?: number + /** Standbys whose WAL receiver is up (cnpg_pg_replication_is_wal_receiver_up = 1). */ + receiving?: number + /** Standbys whose WAL receiver is down: their lag reads 0 because nothing arrives. */ + receiverDown?: string[] + /** The exporter reported no receiver state (or the read failed), so streaming is not established from lag alone. */ + receiverUnknown?: boolean + receiverReason?: string + /** Standbys whose receiver was down in every sample over `receiverDownWindow`: only these raise a problem, since a restarting standby is briefly down. */ + receiverDownSustained?: string[] + receiverDownWindow?: string + } + /** Inactive physical replication slots and the WAL each keeps: ok | noSeries | denied | ambiguous | scopeMismatch | error | notRead */ + slots?: { + state: string + grant?: Grant + reason?: string + /** Null unless `state` is ok; [] when ok and none is inactive. */ + inactive?: { slot: string; pod: string; role?: 'primary' | 'standby'; bytes: number | null }[] | null + /** Inactive slots beyond the per-cluster cap. */ + omitted?: number + isolation?: CNPGMetricIsolation + } + /** ok | noSeries | denied | unavailable | error | notRead */ + growth: { state: string; grant?: Grant; reason?: string; bytesPerHour?: number; claim?: string; instance?: string; isolation?: CNPGMetricIsolation } +} + +/** How Prometheus series were tied to one cluster; `unverified` matched only by namespace and Pod or claim names. */ +export interface CNPGMetricIsolation { + mode: 'configured' | 'verified' | 'unverified' + note: string +} + +// A finding stated as this cluster's must say when its series were matched by name alone. +function isolationCaveat(iso: CNPGMetricIsolation | undefined): string { + return iso?.mode === 'unverified' ? ` ${iso.note}.` : '' +} + +export interface CNPGFleetMetricsSources { + /** prometheus, or none when Radar has no Prometheus (`reason` says why). */ + source: 'prometheus' | 'none' + reason?: string + lagSource?: string + growthSource?: string +} + +export function cnpgLagTone(seconds: number): HealthLevel { + if (seconds >= 30) return 'unhealthy' + if (seconds >= 5) return 'degraded' + return 'healthy' +} + +/** + * Replication's tone from the primary's pg_stat_replication: a missing + * standby is degraded, and the lag of the ones that do stream can make it + * worse. Missing standbys never hide a severe lag. + */ +export function cnpgReplicationTone(streaming: number, expected: number, maxLagSeconds: number | undefined): HealthLevel { + const missing: HealthLevel = streaming < expected ? 'degraded' : 'healthy' + return maxLagSeconds === undefined ? missing : worseTone(missing, cnpgLagTone(maxLagSeconds)) +} + +/** + * The one way a replay lag reads: milliseconds below a second, one decimal + * below 10 s, whole seconds below 100 s, then whole minutes, then hours and + * minutes. Rounded down, so a lower bound stays one. + */ +export function cnpgFormatLag(s: number): string { + if (s <= 0) return '0 s' + if (s < 1) return `${Math.floor(s * 1000)} ms` + if (s < 10) return `${(Math.floor(s * 10) / 10).toFixed(1)} s` + if (s < 100) return `${Math.floor(s)} s` + const minutes = Math.floor(s / 60) + if (minutes < 60) return `${minutes} min` + const m = minutes % 60 + return m === 0 ? `${Math.floor(minutes / 60)} h` : `${Math.floor(minutes / 60)} h ${m} min` +} + +function measuredReplication(base: Fact, reading: CNPGFleetMetricsReading | undefined, src: CNPGFleetMetricsSources, designated?: string, expectedStandbys?: number | null): Fact { + const prefix = base.text.replace(/ · lag unknown$/, '') + if (src.source === 'none') { + return { text: `${prefix} · lag unknown`, tone: 'unknown', source: CNPG_PROMETHEUS_NOT_CONNECTED, detail: src.reason } + } + const lag = reading?.lag + switch (lag?.state) { + case 'ok': { + const down = (lag.receiverDown ?? []).filter((pod) => pod !== designated) + if (down.length > 0) { + return { + text: `${prefix} · ${down.length === 1 ? `${down[0]} not receiving WAL` : `${down.length} standbys not receiving WAL`}`, + tone: lag.receiving === 0 && down.length === lag.standbys ? 'unhealthy' : 'degraded', + source: `WAL receiver down on ${down.join(', ')} (cnpg_pg_replication_is_wal_receiver_up = 0) · ${src.lagSource ?? 'Prometheus'}.${isolationCaveat(lag.isolation)}`, + } + } + if (lag.seconds === undefined) break + // The lag covers the standbys whose lag was read; one that is not read is unknown, not caught up. + const partial = expectedStandbys !== undefined && expectedStandbys !== null && lag.lagStandbys !== undefined && lag.lagStandbys < expectedStandbys + const coverage = partial ? ` (${lag.lagStandbys} of ${expectedStandbys} ${designated ? 'instances' : 'standbys'} reporting)` : '' + const lagTone = cnpgLagTone(lag.seconds) + return { + text: `${prefix} · lag ${cnpgFormatLag(lag.seconds)}${coverage}${lag.receiverUnknown ? ' · streaming unverified' : ''}`, + tone: lag.receiverUnknown || (partial && lagTone === 'healthy') ? 'unknown' : lagTone, + source: `Largest standby replay lag, ${lag.pod ?? 'a standby'} · ${src.lagSource ?? 'Prometheus'}.${lag.receiverUnknown ? ' The exporter reported no WAL receiver state, and a standby that receives nothing also reads 0.' : ''}${isolationCaveat(lag.isolation)}`, + } + } + case 'noStandby': + return { text: `${prefix} · lag unknown`, tone: 'unknown', source: `No standby reports lag: ${lag.reason ?? 'no instance reports being a standby'} · ${src.lagSource ?? 'Prometheus'}` } + case 'denied': + return { text: `${prefix} · lag unknown`, tone: 'unknown', source: lag.grant ? `Needs ${formatGrant(lag.grant)}` : lag.reason } + } + return { text: `${prefix} · lag unknown`, tone: 'unknown', source: lag?.reason ?? 'Replication lag needs Prometheus scraping the CNPG exporter' } +} + +/** Volume growth of the fastest-growing claim, as a fact. */ +export function cnpgDiskGrowthFact(reading: CNPGFleetMetricsReading | undefined, src: CNPGFleetMetricsSources): Fact | undefined { + const g = reading?.growth + if (src.source === 'none' || !g || g.state !== 'ok' || g.bytesPerHour === undefined) return undefined + const perDay = g.bytesPerHour * 24 + const text = Math.abs(perDay) < 1024 ? 'flat over 6 h' : `${perDay > 0 ? '+' : '−'}${formatBytes(Math.abs(perDay))}/day` + return { text, tone: 'neutral', source: `Fastest-growing: ${g.claim ?? 'a volume'}${g.instance ? ` of ${g.instance}` : ''} · ${src.growthSource ?? 'Prometheus'}.${isolationCaveat(g.isolation)}` } +} + +/** + * Joins /api/cnpg/fleet-metrics into the fleet: a cluster whose replication + * fact is only Pod readiness gets its measured standby lag, or says why it has + * none; the disk growth, when measured, lands on `diskGrowth`. Only lag that + * stayed high for the whole sustained window raises a problem; a spike is + * shown, not judged, and growth is never judged. + */ +export function applyCNPGFleetMetrics(fleet: CNPGFleet, readings: CNPGFleetMetricsReading[] | undefined, src: CNPGFleetMetricsSources | undefined): CNPGFleet { + if (!src) return fleet + const byKey = new Map((readings ?? []).map((r) => [key(r.namespace, r.name), r])) + const rows = fleet.rows.map((row) => { + const reading = byKey.get(row.key) + const next: CNPGFleetRow = { ...row, diskGrowth: cnpgDiskGrowthFact(reading, src) } + if (row.replication.source === CNPG_LAG_UNMEASURED_SOURCE) next.replication = measuredReplication( + row.replication, + reading, + src, + row.replicaCluster ? row.cluster?.status?.currentPrimary : undefined, + // Every instance of a replica cluster is in recovery, its designated primary included. + row.instances.desired !== null ? (row.replicaCluster ? row.instances.desired : Math.max(0, row.instances.desired - 1)) : null, + ) + const sustained = sustainedLagProblem(row, reading, src) + return cnpgWithProblems(next, [...(sustained ? [sustained] : []), ...fleetStandbyProblems(row, reading, src), ...fleetSlotProblems(row, reading, src)]) + }) + return finishFleet(rows, fleet.incompleteKinds) +} + +function fleetStandbyProblems(row: CNPGFleetRow, reading: CNPGFleetMetricsReading | undefined, src: CNPGFleetMetricsSources): CNPGProblem[] { + const lag = reading?.lag + if (src.source !== 'prometheus' || lag?.state !== 'ok' || row.hibernated) return [] + // A replica cluster's designated primary is in recovery too, and may be fed + // from the WAL archive with no receiver at all. + const designated = row.replicaCluster ? row.cluster?.status?.currentPrimary : undefined + const down = (lag.receiverDownSustained ?? []).filter((pod) => pod !== designated) + const window = lag.receiverDownWindow ? formatWindowWords(lag.receiverDownWindow) : 'several minutes' + return down.map((pod) => + cnpgStandbyNotReceivingProblem(row, { + pod, + evidence: [`its WAL receiver was down in every sample Prometheus recorded over the last ${window}`], + measuredBy: lag.isolation?.mode === 'unverified' ? 'Prometheus, matched by Pod name' : 'Prometheus', + sourceDetail: `cnpg_pg_replication_is_wal_receiver_up = 0 while cnpg_pg_replication_in_recovery = 1 · ${src.lagSource ?? 'Prometheus'}`, + noneReceiving: noStandbyReceives(row, lag), + unverifiedMatch: lag.isolation?.mode === 'unverified', + }), + ) +} + +// "None receives" only when every expected standby reported its receiver and +// every one was down; one that did not report leaves it open. +function noStandbyReceives(row: CNPGFleetRow, lag: CNPGFleetMetricsReading['lag']): boolean { + const expected = row.instances.desired !== null ? Math.max(0, row.instances.desired - 1) : null + const down = lag.receiverDownSustained?.length ?? 0 + return lag.receiving === 0 && expected !== null && expected > 0 && lag.standbys === expected && down === expected +} + +function fleetSlotProblems(row: CNPGFleetRow, reading: CNPGFleetMetricsReading | undefined, src: CNPGFleetMetricsSources): CNPGProblem[] { + const slots = reading?.slots + if (src.source !== 'prometheus' || slots?.state !== 'ok' || !slots.inactive) return [] + // CloudNativePG copies HA slots to the standbys, where nothing streams from + // them, so a standby's copy is always inactive. Only the primary's copy + // means a consumer stopped. + const instances = row.pods.map((p) => p.name) + return slots.inactive + .filter((s): s is typeof s & { bytes: number } => s.role === 'primary' && s.bytes !== null && s.bytes >= CNPG_SLOT_RETENTION_WARNING_BYTES) + .map((s) => + cnpgSlotRetentionProblem(row, { + slot: s.slot, + pod: s.pod, + bytes: s.bytes, + standby: cnpgHASlotInstance(row.cluster, s.slot, instances), + measuredBy: slots.isolation?.mode === 'unverified' ? 'Prometheus, matched by Pod name' : 'Prometheus', + sourceDetail: 'cnpg_pg_replication_slots_pg_wal_lsn_diff where cnpg_pg_replication_slots_active = 0 (physical slots)', + unverifiedMatch: slots.isolation?.mode === 'unverified', + }), + ) +} + +export const CNPG_SUSTAINED_LAG_WARNING_SECONDS = 30 +export const CNPG_SUSTAINED_LAG_CRITICAL_SECONDS = 300 + +/** The id of a cluster's sustained-lag problem, so other views can find it among the row's problems. */ +export function cnpgSustainedLagProblemId(rowKey: string): string { + return `lag:${rowKey}` +} + +function sustainedLagProblem(row: CNPGFleetRow, reading: CNPGFleetMetricsReading | undefined, src: CNPGFleetMetricsSources): CNPGProblem | undefined { + const lag = reading?.lag + const floor = lag?.sustainedSeconds + if (src.source !== 'prometheus' || lag?.state !== 'ok' || floor === undefined || floor < CNPG_SUSTAINED_LAG_WARNING_SECONDS) return undefined + const window = lag.sustainedWindow ? formatWindowWords(lag.sustainedWindow) : 'several minutes' + const pod = lag.sustainedPod ?? 'A standby' + // The query proves every recorded sample was at least the floor and that + // the series existed at the window's start, not that samples were continuous. + return { + id: cnpgSustainedLagProblemId(row.key), + reason: 'CNPGSustainedLag', + severity: floor >= CNPG_SUSTAINED_LAG_CRITICAL_SECONDS ? 'critical' : 'warning', + category: 'availability', + title: `${pod} ≥ ${cnpgFormatLag(floor)} behind in every sample for ${formatWindowShort(lag.sustainedWindow)}`, + shortTitle: `${pod}: all samples ≥ ${cnpgFormatLag(floor)} behind (${formatWindowShort(lag.sustainedWindow)})`, + detail: `Lowest replay lag in the samples Prometheus recorded over the last ${window}. If Prometheus missed some scrapes, those moments aren't included. If it was still that far behind, a failover to it would lose or wait on that much WAL.${isolationCaveat(lag.isolation)}`, + subject: { kind: 'Cluster', group: 'postgresql.cnpg.io', namespace: row.namespace, name: row.name }, + source: 'measurement', + measuredBy: lag.isolation?.mode === 'unverified' ? 'Prometheus, matched by Pod name' : 'Prometheus', + unverifiedMatch: lag.isolation?.mode === 'unverified', + sourceDetail: src.lagSource ?? 'Prometheus', + } +} + +// "10m0s" as "10 min"; "1h0m0s" as "1 h", for a title that must stay short. +function formatWindowShort(d: string | undefined): string { + const m = d ? /^(?:(\d+)h)?(?:(\d+)m)?(?:0s)?$/.exec(d) : null + if (!m) return d ?? 'minutes' + const minutes = Number(m[1] ?? 0) * 60 + Number(m[2] ?? 0) + return minutes >= 60 && minutes % 60 === 0 ? `${minutes / 60} h` : `${minutes} min` +} + +// "10m0s" as "10 minutes"; "1h0m0s" as "1 hour". +function formatWindowWords(d: string): string { + const m = /^(?:(\d+)h)?(?:(\d+)m)?(?:0s)?$/.exec(d) + if (!m) return d + const minutes = Number(m[1] ?? 0) * 60 + Number(m[2] ?? 0) + if (minutes >= 60 && minutes % 60 === 0) return minutes === 60 ? '1 hour' : `${minutes / 60} hours` + return minutes === 1 ? '1 minute' : `${minutes} minutes` +} diff --git a/packages/k8s-ui/src/components/dock/DockContext.tsx b/packages/k8s-ui/src/components/dock/DockContext.tsx index 02ee2083f6..b7fbb76ee4 100644 --- a/packages/k8s-ui/src/components/dock/DockContext.tsx +++ b/packages/k8s-ui/src/components/dock/DockContext.tsx @@ -16,6 +16,10 @@ export interface DockTab { podName?: string containerName?: string containers?: string[] + /** Terminal: run this command instead of a shell (one argv element, e.g. "psql"). */ + shell?: string + /** Terminal: a short line shown in the toolbar, e.g. what the session is connected to. */ + sessionNote?: string // Workload logs props workloadKind?: string workloadName?: string @@ -87,7 +91,8 @@ export function DockProvider({ children }: { children: ReactNode }) { } return t.namespace === tabData.namespace && t.podName === tabData.podName && - t.containerName === tabData.containerName + t.containerName === tabData.containerName && + t.shell === tabData.shell }) if (existingTab) { @@ -207,14 +212,19 @@ export function useOpenTerminal() { orgId?: string clusterId?: string clusterName?: string + shell?: string + sessionNote?: string + title?: string }) => { addTab({ type: 'terminal', - title: `${opts.podName}/${opts.containerName}`, + title: opts.title ?? `${opts.podName}/${opts.containerName}`, namespace: opts.namespace, podName: opts.podName, containerName: opts.containerName, containers: opts.containers, + shell: opts.shell, + sessionNote: opts.sessionNote, orgId: opts.orgId, clusterId: opts.clusterId, clusterName: opts.clusterName, diff --git a/packages/k8s-ui/src/components/dock/TerminalTab.tsx b/packages/k8s-ui/src/components/dock/TerminalTab.tsx index 68cedf5abf..f1d06f0192 100644 --- a/packages/k8s-ui/src/components/dock/TerminalTab.tsx +++ b/packages/k8s-ui/src/components/dock/TerminalTab.tsx @@ -19,6 +19,8 @@ export interface TerminalTabProps { createSession: (containerName: string) => Promise<{ wsUrl: string }> /** Optional: creates a debug (ephemeral) container. If omitted, the debug button is hidden. */ createDebugContainer?: (targetContainer: string) => Promise<{ containerName: string }> + /** Optional: a short line in the toolbar describing the session. */ + note?: string } export function TerminalTab({ @@ -29,6 +31,7 @@ export function TerminalTab({ isActive = true, createSession, createDebugContainer, + note, }: TerminalTabProps) { const terminalRef = useRef(null) const xtermRef = useRef(null) @@ -269,6 +272,7 @@ export function TerminalTab({ )} /> {podName} + {note && {note}} {containers.length > 1 && (
diff --git a/packages/k8s-ui/src/components/facts/certainty.tsx b/packages/k8s-ui/src/components/facts/certainty.tsx new file mode 100644 index 0000000000..c33d2ffbfb --- /dev/null +++ b/packages/k8s-ui/src/components/facts/certainty.tsx @@ -0,0 +1,36 @@ +import { WithTooltip } from '../ui/Tooltip' + +/** + * How exactly a value is known. A value read only in part is a lower bound, + * never shown as if exact. + */ +export type Certainty = 'exact' | 'lower_bound' | 'upper_bound' | 'unknown' + +export function certaintyGlyph(certainty: Certainty): string { + if (certainty === 'exact') return '=' + if (certainty === 'lower_bound') return '≥' + if (certainty === 'upper_bound') return '≤' + return '?' +} + +export function certaintyValueLabel(certainty: Certainty): string { + if (certainty === 'exact') return 'Exact' + if (certainty === 'lower_bound') return 'Lower bound' + if (certainty === 'upper_bound') return 'Upper bound' + return 'Unknown certainty' +} + +export function CertaintyGlyph({ certainty, title }: { certainty: Certainty; title?: string }) { + return ( + + + {certaintyGlyph(certainty)} + + + ) +} diff --git a/packages/k8s-ui/src/components/facts/facts.test.tsx b/packages/k8s-ui/src/components/facts/facts.test.tsx new file mode 100644 index 0000000000..a560d293f2 --- /dev/null +++ b/packages/k8s-ui/src/components/facts/facts.test.tsx @@ -0,0 +1,16 @@ +import { describe, expect, it } from 'vitest' +import { renderToStaticMarkup } from 'react-dom/server' +import { FactValue } from './facts' + +describe('FactValue', () => { + const twoDaysAgo = new Date(Date.now() - 2 * 24 * 3600 * 1000 - 60_000).toISOString() + it('reads a still-current state as lasting since its timestamp', () => { + const html = renderToStaticMarkup() + expect(html).toContain('Failing for 2d') + expect(html).not.toContain('ago') + }) + it('reads a past event as an age', () => { + const html = renderToStaticMarkup() + expect(html).toContain('2d ago') + }) +}) diff --git a/packages/k8s-ui/src/components/facts/facts.tsx b/packages/k8s-ui/src/components/facts/facts.tsx new file mode 100644 index 0000000000..3b6409d530 --- /dev/null +++ b/packages/k8s-ui/src/components/facts/facts.tsx @@ -0,0 +1,60 @@ +import type { ReactNode } from 'react' +import { clsx } from 'clsx' +import type { HealthLevel } from '../resources/resource-utils' +import { formatAge } from '../resources/resource-utils' +import { toneTextClass } from '../ui/status-tone' +import { Tooltip } from '../ui/Tooltip' + +/** + * One observed value and where it came from. A value the cluster does not + * report is a fact too: its text says so and its tone is `unknown`, never a + * zero or a calm default. + */ +export interface Fact { + text: string + tone: HealthLevel + /** Where the value comes from, shown next to it so claims carry their source. */ + source?: string + /** A timestamp the text refers to; the UI renders it as an age. */ + at?: string + /** `since`: `at` is when a still-current state began, rendered "Failing for 2d" rather than "· 2d ago". */ + atMeaning?: 'since' + /** The full explanation behind a short `source`, shown on hover only. */ + detail?: string +} + +export function FactValue({ fact, className }: { fact: Fact; className?: string }) { + const age = fact.at ? formatAge(fact.at) : null + const body = ( + + {fact.text} + {age && fact.atMeaning === 'since' && for {age}} + {age && fact.atMeaning !== 'since' && {fact.text ? ' · ' : ''}{age} ago} + + ) + if (!fact.source && !fact.at && !fact.detail) return body + return ( + + {body} + + ) +} + +export function FactSource({ fact }: { fact: Fact }) { + if (!fact.source) return null + return
{fact.source}
+} + +/** Label/value rows. Empty values stay visible: an unread value is shown as unread, not hidden. */ +export function FactGrid({ children }: { children: ReactNode }) { + return
{children}
+} + +export function FactRow({ label, children }: { label: ReactNode; children: ReactNode }) { + return ( + <> +
{label}
+
{children}
+ + ) +} diff --git a/packages/k8s-ui/src/components/facts/index.ts b/packages/k8s-ui/src/components/facts/index.ts new file mode 100644 index 0000000000..ce2f304a7c --- /dev/null +++ b/packages/k8s-ui/src/components/facts/index.ts @@ -0,0 +1,6 @@ +// How a surface shows what the cluster reported: each value with its source, +// partial and unread values marked as such (see DESIGN.md, "Unknown, partial +// and denied values"). For any renderer, summary or workspace screen. +export * from './facts' +export * from './certainty' +export * from './managed-by' diff --git a/packages/k8s-ui/src/components/facts/managed-by.test.tsx b/packages/k8s-ui/src/components/facts/managed-by.test.tsx new file mode 100644 index 0000000000..842c51f1ef --- /dev/null +++ b/packages/k8s-ui/src/components/facts/managed-by.test.tsx @@ -0,0 +1,26 @@ +import { describe, expect, it } from 'vitest' +import { renderToStaticMarkup } from 'react-dom/server' +import { ManagedByText, managedByLabel } from './managed-by' +import { cnpgManagedBy } from '../cnpg/workspace' + +describe('ManagedByText', () => { + it('names the GitOps manager and links it when its namespace is recorded', () => { + const app = { kind: 'Application', group: 'argoproj.io', namespace: 'argocd', name: 'payments' } + const html = renderToStaticMarkup( {}} />) + expect(html).toContain('Argo CD application') + expect(html).toMatch(/]*>argocd\/payments<\/button>/) + expect(renderToStaticMarkup( {}} />)).not.toContain(' { + expect(renderToStaticMarkup()).toBe('') + expect(managedByLabel({ kind: 'Kustomization', group: 'kustomize.toolkit.fluxcd.io', namespace: 'flux-system', name: 'apps' })).toBe('Flux Kustomization flux-system/apps') + }) +}) + +describe('cnpgManagedBy', () => { + it('looks an object up by kind, namespace and name in the workspace answer', () => { + const ws = { managedBy: { 'Cluster/db/pg': { kind: 'Application', group: 'argoproj.io', namespace: 'argocd', name: 'pg' } } } + expect(cnpgManagedBy(ws, { kind: 'Cluster', metadata: { namespace: 'db', name: 'pg' } })?.name).toBe('pg') + expect(cnpgManagedBy(ws, { kind: 'Pooler', metadata: { namespace: 'db', name: 'pg' } })).toBeUndefined() + }) +}) diff --git a/packages/k8s-ui/src/components/facts/managed-by.tsx b/packages/k8s-ui/src/components/facts/managed-by.tsx new file mode 100644 index 0000000000..c5ba1c1ec4 --- /dev/null +++ b/packages/k8s-ui/src/components/facts/managed-by.tsx @@ -0,0 +1,36 @@ +import type { ResourceRef } from '../../types/core' +import { gitOpsOwnerFromRef, type GitOpsOwnerRef } from '../../utils/gitops-owner' +import { RefLink, type NavigateToRef } from '../ui/RefLink' + +function managerLabel(owner: GitOpsOwnerRef): string { + if (owner.tool === 'argocd') return 'Argo CD application' + return owner.kind === 'helmreleases' ? 'Flux HelmRelease' : 'Flux Kustomization' +} + +/** + * The GitOps object that manages a resource, from the server's manager + * detection. Renders nothing for a manager that is not a GitOps controller, + * and plain text when the manager's namespace is not recorded. + */ +export function ManagedByText({ refTo, onNavigate }: { refTo: ResourceRef; onNavigate?: NavigateToRef }) { + const owner = gitOpsOwnerFromRef(refTo) + if (!owner) return null + return ( + + {managerLabel(owner)}{' '} + {refTo.namespace ? ( + + {`${refTo.namespace}/${refTo.name}`} + + ) : ( + {refTo.name} + )} + + ) +} + +/** The manager's label and name as text, for a table cell. */ +export function managedByLabel(refTo: ResourceRef | undefined): string | undefined { + const owner = refTo && gitOpsOwnerFromRef(refTo) + return owner ? `${managerLabel(owner)} ${refTo!.namespace ? `${refTo!.namespace}/` : ''}${refTo!.name}` : undefined +} diff --git a/packages/k8s-ui/src/components/gitops/detail-helpers.ts b/packages/k8s-ui/src/components/gitops/detail-helpers.ts index 1956312321..761dfd6a50 100644 --- a/packages/k8s-ui/src/components/gitops/detail-helpers.ts +++ b/packages/k8s-ui/src/components/gitops/detail-helpers.ts @@ -1,3 +1,4 @@ +import { stripTrailingSlashes } from '../../utils/url-path' import { argoApplicationSetConditionsToGitOpsStatus, argoStatusToGitOpsStatus, fluxConditionsToGitOpsStatus, type FluxCondition, type GitOpsStatus } from '../../types/gitops' import type { GitOpsResourceTree } from '../../types/gitops-tree' import { formatCompactAge } from '../../utils/format' @@ -33,7 +34,7 @@ export function formatGitOpsDestination(server: string | undefined, namespace: s // "https://kubernetes.default.svc/" variant (the in-cluster URL with a // trailing slash that some controller versions emit) matches the literal // we collapse to "in-cluster". - let host = (server || '').trim().replace(/\/+$/, '') + let host = stripTrailingSlashes((server || '').trim()) if (host === '' || host === 'https://kubernetes.default.svc' || host === 'in-cluster') { host = 'in-cluster' } else { diff --git a/packages/k8s-ui/src/components/issues/IssuesView.tsx b/packages/k8s-ui/src/components/issues/IssuesView.tsx index 01384d67d4..ae54fbdb3f 100644 --- a/packages/k8s-ui/src/components/issues/IssuesView.tsx +++ b/packages/k8s-ui/src/components/issues/IssuesView.tsx @@ -1,3 +1,4 @@ +import { summarizeSchedulerMessage } from '../resources/resource-utils' import { useMemo, useState, type ComponentType, type ReactNode } from 'react'; import { AlertOctagon, AlertTriangle, ArrowRight, CircleCheck, Clock, ExternalLink, Layers, Terminal, Workflow } from 'lucide-react'; import { CardBody, CardSection, ClusterName, EmptyState, KIND_CHIP_CLASS, TerminalBlock } from '../ui'; @@ -13,7 +14,7 @@ import { ISSUE_SEVERITY_RAIL_CLASS, ISSUE_SEVERITY_SOLID_CLASS, ISSUE_SEVERITY_TEXT_CLASS, - categoryLabel, + issueTitle, groupBadgeClass, groupLabel, } from './severity'; @@ -168,6 +169,7 @@ export interface IssueRowProps { resourceHref?: (ref: IssueResourceRef) => string; onResourceClick?: (ref: IssueResourceRef) => void; as?: 'li' | 'div'; + compact?: boolean; className?: string; dimmed?: boolean; /** Suppress the "Subject" deep-link in the expanded body — set by hosts that @@ -195,6 +197,7 @@ export function IssueRow({ resourceHref, onResourceClick, as = 'li', + compact = false, className, dimmed, hideSubject, @@ -208,6 +211,7 @@ export function IssueRow({ const cluster = clusterLabel?.(issue); const affected = affectedSummary(issue.affected); const { headline } = issueMessageParts(issue); + const schedulingCause = compact && issue.reason === 'Unschedulable' ? summarizeSchedulerMessage(issue.cause || issue.message || headline || '', { plain: true }) : undefined; const { panelId, buttonProps } = useDisclosure(open); const Container = as; const severity = normalizeIssueSeverity(issue.severity); @@ -289,7 +293,7 @@ export function IssueRow({
- {categoryLabel(issue.category)} + {issueTitle(issue)} {groupLabel(issue.category_group)} {renderBadges?.(slotCtx)} {/* The detector reason rides the title row while COLLAPSED so the @@ -297,7 +301,7 @@ export function IssueRow({ cause lives in the WHAT'S WRONG section below, so it fades out here (stays mounted — it's the flex-1 filler, so unmounting wouldn't reflow anything, but fading avoids the pop). */} - {issue.reason ? ( + {issue.reason && !schedulingCause ? ( ) : null}
+ {schedulingCause &&
{schedulingCause}
}
{issue.kind} diff --git a/packages/k8s-ui/src/components/issues/ResourceIssuesSection.test.tsx b/packages/k8s-ui/src/components/issues/ResourceIssuesSection.test.tsx index 5700d3bc7a..4849556cdf 100644 --- a/packages/k8s-ui/src/components/issues/ResourceIssuesSection.test.tsx +++ b/packages/k8s-ui/src/components/issues/ResourceIssuesSection.test.tsx @@ -106,3 +106,31 @@ it('keeps pod/template evidence in expanded neutral context with navigable witne expect(expanded).not.toContain('confidence') expect(expanded).not.toContain('Caused by') }) + +it('wraps a compact Pod scheduling cause separately and retains the full message in details', () => { + const raw = '0/2 nodes are available: 2 Too many pods. preemption: 0/2 nodes are available: 2 No preemption victims found for incoming pod.' + const pod: Issue = { ...issue, kind: 'Pod', reason: 'Unschedulable', cause: raw, message: raw } + const html = renderToString( {}} />) + expect(html).toContain('break-words text-xs text-theme-text-secondary') + expect(html).toContain('both nodes have reached their Pod limit') + expect(html).toContain(raw) + const producerMessage = '2 node(s) insufficient pods (0/2 nodes available)' + const reported = renderToString( {}} />) + expect(reported).toContain('both nodes have reached their Pod limit') + expect(reported).toContain(producerMessage) + const regular = renderToString( {}} />) + expect(regular).not.toContain('both nodes have reached their Pod limit') +}) + +it.each(['Cluster', 'Deployment'])('wraps compact scheduling causes aggregated under %s', (kind) => { + const raw = '2 node(s) insufficient pods (0/2 nodes available)' + const aggregated: Issue = { ...issue, kind, reason: 'Unschedulable', cause: raw, message: raw } + const collapsed = renderToString( {}} />) + expect(collapsed).toContain('
both nodes have reached their Pod limit
') + expect(collapsed).not.toContain(raw) + const expanded = renderToString( {}} />) + expect(expanded).toContain(raw) + const regular = renderToString( {}} />) + expect(regular).toContain('Unschedulable') + expect(regular).not.toContain('both nodes have reached their Pod limit') +}) diff --git a/packages/k8s-ui/src/components/issues/ResourceIssuesSection.tsx b/packages/k8s-ui/src/components/issues/ResourceIssuesSection.tsx index d80e58fb18..074f9fffdc 100644 --- a/packages/k8s-ui/src/components/issues/ResourceIssuesSection.tsx +++ b/packages/k8s-ui/src/components/issues/ResourceIssuesSection.tsx @@ -8,7 +8,9 @@ export function ResourceIssuesSection({ issues, onResourceClick, subjectResource, + compact, }: { + compact?: boolean issues: Issue[] | undefined /** When provided, related resources in a causal link become clickable. */ onResourceClick?: (ref: IssueResourceRef) => void @@ -36,6 +38,7 @@ export function ResourceIssuesSection({ setOpenId((cur) => (cur === key ? null : key))} onResourceClick={onResourceClick} diff --git a/packages/k8s-ui/src/components/issues/index.ts b/packages/k8s-ui/src/components/issues/index.ts index 48260f5235..93fcb849b1 100644 --- a/packages/k8s-ui/src/components/issues/index.ts +++ b/packages/k8s-ui/src/components/issues/index.ts @@ -24,5 +24,7 @@ export { ISSUE_SEVERITY_FILL_CLASS, ISSUE_SEVERITY_RAIL_CLASS, categoryLabel, + issueTitle, + issueReasonTitle, groupLabel, } from './severity'; diff --git a/packages/k8s-ui/src/components/issues/issues.test.ts b/packages/k8s-ui/src/components/issues/issues.test.ts index f31625e481..8311602cc5 100644 --- a/packages/k8s-ui/src/components/issues/issues.test.ts +++ b/packages/k8s-ui/src/components/issues/issues.test.ts @@ -2,7 +2,7 @@ import { afterEach, describe, it, expect, vi } from 'vitest' import { createElement } from 'react' import { renderToString } from 'react-dom/server' import { compareIssues, issueSortAnchor, subjectRef, memberRef, normalizeImagePullMessage, issueMessageParts, type Issue } from './types' -import { categoryLabel, groupBadgeClass, groupLabel } from './severity' +import { categoryLabel, groupBadgeClass, groupLabel, issueTitle } from './severity' import { IssueRow } from './IssuesView' import { issueFirstSeenTitle, issueResourceCreatedTitle, issueTiming } from './issue-timing' @@ -83,6 +83,8 @@ describe('category/group label fallbacks', () => { it('returns the mapped label, else humanizes (server-added category needs no frontend deploy)', () => { expect(categoryLabel('crashloop')).toBe('Crash loop') expect(categoryLabel('some_new_future_category')).toBe('Some new future category') + expect(issueTitle({ category: 'backup_failed', reason: 'CNPGWALArchivingFailing' })).toBe('WAL archiving failing') + expect(issueTitle({ category: 'backup_failed', reason: 'BackupFailed' })).toBe('Backup failed') }) it('humanizes an unmapped group', () => { expect(groupLabel('runtime')).toBe('Runtime') diff --git a/packages/k8s-ui/src/components/issues/severity.ts b/packages/k8s-ui/src/components/issues/severity.ts index 9272169213..6417e9a185 100644 --- a/packages/k8s-ui/src/components/issues/severity.ts +++ b/packages/k8s-ui/src/components/issues/severity.ts @@ -136,6 +136,27 @@ export function categoryLabel(category: string): string { return CATEGORY_LABEL[category] ?? humanize(category); } +// Reasons filed under a category whose label names a different operation: +// CNPG archiving and an unanswered scheduled run sit under backup_failed. +const REASON_TITLE: Record = { + CNPGWALArchivingFailing: "WAL archiving failing", + CNPGLastBackupFailed: "Latest backup failed", + CNPGScheduleDestinationMissing: "Backup schedule has no destination", + CNPGInstanceReadinessMismatch: "Instance Pods contradict Cluster readiness", + CNPGPrimaryLabelMismatch: "Primary Pod label contradicts Cluster status", + CNPGScheduledRunNoBackup: "No successful backup since a scheduled run", +}; + +/** The title a reason has wherever it is shown, when its category's label would name the wrong operation. */ +export function issueReasonTitle(reason: string | undefined): string | undefined { + return reason ? REASON_TITLE[reason] : undefined; +} + +/** An issue row's title: its category, unless the reason names it better. */ +export function issueTitle(issue: { category: string; reason?: string }): string { + return issueReasonTitle(issue.reason) || categoryLabel(issue.category); +} + export function groupLabel(group: string): string { return GROUP_LABEL[group] ?? humanize(group); } diff --git a/packages/k8s-ui/src/components/logs/LogCore.theme.test.tsx b/packages/k8s-ui/src/components/logs/LogCore.theme.test.tsx index 2affc30e14..8c6ac30789 100644 --- a/packages/k8s-ui/src/components/logs/LogCore.theme.test.tsx +++ b/packages/k8s-ui/src/components/logs/LogCore.theme.test.tsx @@ -1,9 +1,11 @@ // @vitest-environment jsdom -import { act } from 'react' +import { act, type ComponentProps, type ReactNode } from 'react' import { createRoot, type Root } from 'react-dom/client' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { LogCore } from './LogCore' +vi.mock('react-virtuoso', () => ({ Virtuoso: ({ data, itemContent }: { data: unknown[]; itemContent: (index: number, value: unknown) => ReactNode }) =>
{data.map((value, index) =>
{itemContent(index, value)}
)}
})) + Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true }) let root: Root @@ -20,7 +22,7 @@ afterEach(async () => { }) const noop = () => {} -async function render(props: { forceDark?: boolean; defaultDark?: boolean }) { +async function render(props: Partial>) { await act(async () => root.render( { expect(toggle()).toBeNull() }) }) + +it('dims unavailable streaming, explains why, suppresses its hint and keyboard action', async () => { + vi.useFakeTimers() + const start = vi.fn() + await render({ onStartStream: start, sourceUnavailable: true }) + const stream = [...element.querySelectorAll('button')].find((b) => b.textContent === 'Stream')! + expect(stream.disabled).toBe(true) + expect(stream.classList.contains('disabled:opacity-50')).toBe(true) + expect([...element.querySelectorAll('kbd')].some((k) => k.textContent === 'S')).toBe(false) + act(() => { stream.parentElement!.dispatchEvent(new MouseEvent('mouseover', { bubbles: true })); vi.advanceTimersByTime(500) }) + expect(document.body.textContent).toContain('Available once an instance is running') + act(() => window.dispatchEvent(new KeyboardEvent('keydown', { key: 's' }))) + expect(start).not.toHaveBeenCalled() + vi.useRealTimers() +}) +it('keeps a non-CNPG caller streamable without Pods and wraps its utilities together', async () => { + const start = vi.fn() + await render({ onStartStream: start, onClear: noop, toolbarExtra: Deployment nginx sources, entries: [{ id: 1, content: '{"msg":"nginx started","level":"info"}', timestamp: '', container: 'nginx', pod: 'nginx-1', isJson: true, isLogfmt: false, level: 'info', levelSource: 'structured' }] }) + const stream = [...element.querySelectorAll('button')].find((b) => b.textContent === 'Stream')! + expect(stream.disabled).toBe(false) + expect([...element.querySelectorAll('kbd')].some((k) => k.textContent === 'S')).toBe(true) + act(() => stream.click()) + expect(start).toHaveBeenCalledOnce() + const utilities = element.querySelector('[aria-label="Log display and utilities"]')! + expect(utilities.classList.contains('flex-nowrap')).toBe(true) + expect(utilities.classList.contains('shrink-0')).toBe(true) + for (const icon of ['lucide-trash-2', 'lucide-download', 'lucide-search', 'lucide-clock', 'lucide-braces']) expect(utilities.querySelector(`.${icon}`)).not.toBeNull() +}) + +it('still stops an existing stream when its source becomes unavailable', async () => { + const stop = vi.fn() + await render({ onStartStream: noop, onStopStream: stop, isStreaming: true, sourceUnavailable: true }) + const button = [...element.querySelectorAll('button')].find((b) => b.textContent === 'Stop')! + expect(button.disabled).toBe(false) + act(() => window.dispatchEvent(new KeyboardEvent('keydown', { key: 's' }))) + expect(stop).toHaveBeenCalledOnce() +}) + +it.each([true, false])('wraps non-CNPG plain logs at words with ANSI enabled=%s', async (ansi) => { + localStorage.setItem('radar-logs-ansi', String(ansi)) + await render({ entries: [{ id: 1, timestamp: '', content: 'nginx serving requests', container: 'nginx', isJson: false, isLogfmt: false, level: 'info', levelSource: 'keyword' }] }) + const line = [...element.querySelectorAll('span')].find((el) => el.textContent === 'nginx serving requests' && el.classList.contains('whitespace-pre-wrap'))! + expect(line.classList.contains('[overflow-wrap:anywhere]')).toBe(true) + expect(line.classList.contains('break-all')).toBe(false) +}) +it('keeps normal word wrapping when searching a structured log', async () => { + await render({ entries: [{ id: 1, timestamp: '', content: '{"msg":"nginx serving requests"}', container: 'nginx', isJson: true, isLogfmt: false, level: 'info', levelSource: 'structured' }] }) + await act(async () => window.dispatchEvent(new KeyboardEvent('keydown', { key: 'f', ctrlKey: true }))) + const input = element.querySelector('input[placeholder="Search logs..."]')! + await act(async () => { + Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, 'value')!.set!.call(input, 'nginx') + input.dispatchEvent(new Event('input', { bubbles: true })) + }) + const highlight = element.querySelector('mark')! + expect(highlight).not.toBeNull() + expect(highlight.parentElement!.classList.contains('[overflow-wrap:anywhere]')).toBe(true) + expect(highlight.parentElement!.classList.contains('break-all')).toBe(false) +}) + +it('disables empty Export and Clear while preserving Refresh and enables them after logs load', async () => { + await render({ onClear: noop }) + const exportButton = () => element.querySelector('[aria-label="Export logs"]')! + const clear = () => element.querySelector('[aria-label="Clear logs"]')! + expect(exportButton().disabled).toBe(true) + expect(clear().disabled).toBe(true) + expect(element.querySelector('[aria-label="Refresh logs"]')?.disabled).toBe(false) + await render({ onClear: noop, entries: [{ id: 1, container: 'postgres', level: 'info', levelSource: 'keyword', isJson: false, isLogfmt: false, content: 'hello', timestamp: '2026-10-01T00:00:00Z' }] }) + expect(exportButton().disabled).toBe(false) + expect(clear().disabled).toBe(false) +}) + +it('keeps Export and Clear usable when a Pod filter hides a nonempty buffer', async () => { + await render({ entries: [], allEntries: [{ id: 1, container: 'postgres', level: 'info', levelSource: 'keyword', isJson: false, isLogfmt: false, content: 'hello', timestamp: '' }], onClear: noop }) + expect(element.querySelector('[aria-label="Export logs"]')!.disabled).toBe(false) + expect(element.querySelector('[aria-label="Clear logs"]')!.disabled).toBe(false) +}) diff --git a/packages/k8s-ui/src/components/logs/LogCore.tsx b/packages/k8s-ui/src/components/logs/LogCore.tsx index 38b753b29a..993d1c35b9 100644 --- a/packages/k8s-ui/src/components/logs/LogCore.tsx +++ b/packages/k8s-ui/src/components/logs/LogCore.tsx @@ -58,7 +58,8 @@ interface LogCoreProps { onClear?: () => void toolbarExtra?: ToolbarExtraRenderer showPodName?: boolean - emptyMessage?: string + emptyMessage?: ReactNode + sourceUnavailable?: boolean emptyCommand?: string | null errorMessage?: string | null /** @@ -146,6 +147,7 @@ export function LogCore({ toolbarExtra, showPodName = false, emptyMessage = 'No logs available', + sourceUnavailable = false, emptyCommand, errorMessage, forceDark, @@ -422,7 +424,7 @@ export function LogCore({ search.open() return } - if (e.key === 's' && !e.ctrlKey && !e.metaKey && !e.altKey && onStartStream) { + if (e.key === 's' && !e.ctrlKey && !e.metaKey && !e.altKey && onStartStream && (isStreaming || !sourceUnavailable)) { const target = e.target as HTMLElement | null if (target && (target.tagName === 'INPUT' || target.tagName === 'TEXTAREA' || target.tagName === 'SELECT' || target.isContentEditable)) return e.preventDefault() @@ -432,7 +434,7 @@ export function LogCore({ } window.addEventListener('keydown', handleKeyDown) return () => window.removeEventListener('keydown', handleKeyDown) - }, [search.open, onStartStream, onStopStream, isStreaming]) + }, [search.open, onStartStream, onStopStream, isStreaming, sourceUnavailable]) const handleFollowOutput = useCallback((isAtBottom: boolean) => { if (isAtBottom) return 'smooth' as const @@ -568,15 +570,16 @@ export function LogCore({ style={{ colorScheme: isDark ? 'dark' : 'light', fontFamily: "'SF Mono', 'Cascadia Code', 'Fira Code', Menlo, Consolas, 'DejaVu Sans Mono', monospace" }} > {/* Toolbar */} -
+
{toolbarExtraNode} {/* Stream / Stop toggle — only shown when streaming is supported */} {onStartStream && ( - +
-
+
{/* Structured-log display mode: icon cycles compact→expanded→raw, chevron picks explicitly. */} {hasStructuredEntries && ( @@ -801,14 +805,15 @@ export function LogCore({ {/* Export */}
- + @@ -884,15 +889,18 @@ export function LogCore({ {/* Clear */} {onClear && ( - + )} +
{/* Search bar */} @@ -1007,7 +1015,7 @@ export function LogCore({ ) : groupedEntries.length === 0 ? (
- {entries.length > 0 ? `Filters hide all ${entries.length.toLocaleString()} loaded lines` : emptyMessage} +
{entries.length > 0 ? `Filters hide all ${entries.length.toLocaleString()} loaded lines` : emptyMessage}
{entries.length === 0 && emptyCommand && ( } + {incompleteReason && !disabledReason && {incompleteReason}} + + +
+ + ) +} diff --git a/packages/k8s-ui/src/components/shared/CreateResourceDialog.test.tsx b/packages/k8s-ui/src/components/shared/CreateResourceDialog.test.tsx new file mode 100644 index 0000000000..3dba3ffe30 --- /dev/null +++ b/packages/k8s-ui/src/components/shared/CreateResourceDialog.test.tsx @@ -0,0 +1,97 @@ +// @vitest-environment jsdom +import { act } from 'react' +import { createRoot } from 'react-dom/client' +import { afterEach, describe, expect, it, vi } from 'vitest' + +vi.mock('../ui/YamlEditor', () => ({ + YamlEditor: ({ value, onChange }: { value: string; onChange: (value: string) => void }) =>