From a80baa43361e4443709e687e95cf141995d15cbe Mon Sep 17 00:00:00 2001 From: Kim Doberstein Date: Fri, 11 Sep 2026 09:48:39 -0500 Subject: [PATCH 1/2] Spec changes to lean toward Prometheus for dashboard --- specs/index.spec.md | 7 +- specs/platform/cluster-cpu.spec.md | 2 +- specs/platform/cluster-memory.spec.md | 6 +- specs/platform/cluster-nodes.spec.md | 2 +- specs/platform/cluster-pods.spec.md | 2 +- .../gateway-metrics-dashboard.spec.md | 6 +- specs/platform/gateway-provision-time.spec.md | 12 +- .../openshell-gateway-sandbox-count.spec.md | 10 ++ specs/platform/platform-inventory.spec.md | 96 ++++++++----- specs/platform/registered-users.spec.md | 38 +++-- .../web-console/operational-dashboard.spec.md | 134 +++++++++++------- 11 files changed, 198 insertions(+), 117 deletions(-) diff --git a/specs/index.spec.md b/specs/index.spec.md index b5f510509..957c4614b 100644 --- a/specs/index.spec.md +++ b/specs/index.spec.md @@ -43,10 +43,11 @@ Machine-readable index for autonomous reconciliation (`/reconcile` skill). | `platform/openshell-inference-routing.spec.md` | platform | Inference router, inference.local, credential-free sandbox model access, provider translation | CP | openshell-gateway, openshell-gateway-credentials | | `platform/global-architecture.spec.md` | platform | Global hub, multi-cloud, CNPG, Tekton, ArgoCD, Vault | CP, ALL | data-model, control-plane | | `web-console/architecture.spec.md` | web-console | Web console, BFF, browser session, UI routes | WEB, SDK, API | data-model, security, openshell-gateway-service-accounts, UI standards | -| `web-console/operational-dashboard.spec.md` | web-console | Widgetized operational dashboard, gateway list metrics adapter, admin access | WEB, SDK | web-console/architecture, gateway-metrics-dashboard, UI standards | +| `web-console/operational-dashboard.spec.md` | web-console | Widgetized operational dashboard, Prometheus-backed metrics adapter, admin access | WEB | web-console/architecture, gateway-metrics-dashboard, platform-inventory, registered-users, UI standards | | `web-console/tracing.spec.md` | web-console | Browser OTel trace sink, BFF W3C propagation, telemetry ingest, dev Jaeger | WEB, BFF | web-console/architecture, domain-observability, local-development | -| `platform/gateway-metrics-dashboard.spec.md` | platform | Prometheus gateway phase metric, BFF metrics proxy, GatewayMetricsDashboard | API, WEB, deploy | data-model, web-console/architecture, local-development | -| `platform/registered-users.spec.md` | platform | Registered user inventory API and operational dashboard count | API, WEB, SDK | rbac-enforcement, web-console/operational-dashboard | +| `platform/gateway-metrics-dashboard.spec.md` | platform | Prometheus gateway phase metric, BFF metrics proxy, GatewayMetricsDashboard, operational dashboard gateway counts | API, WEB, deploy | data-model, web-console/architecture, local-development | +| `platform/platform-inventory.spec.md` | platform | Managed cluster and database inventory Prometheus collectors and operational dashboard widgets | API, WEB | web-console/operational-dashboard, managed-cluster-registration, rbac-enforcement | +| `platform/registered-users.spec.md` | platform | Registered user inventory API, Prometheus count, and operational dashboard widget | API, WEB | rbac-enforcement, web-console/operational-dashboard | | `platform/cluster-memory.spec.md` | platform | Hub cluster memory utilization for operational dashboard | WEB, deploy | web-console/operational-dashboard, gateway-metrics-dashboard, local-development | | `platform/cluster-cpu.spec.md` | platform | Hub cluster CPU utilization for operational dashboard | WEB, deploy | web-console/operational-dashboard, cluster-memory, gateway-metrics-dashboard, local-development | | `platform/cluster-pods.spec.md` | platform | Hub cluster pod utilization for operational dashboard | WEB, deploy | web-console/operational-dashboard, cluster-memory, cluster-cpu, gateway-metrics-dashboard, local-development | diff --git a/specs/platform/cluster-cpu.spec.md b/specs/platform/cluster-cpu.spec.md index 622743599..72ca15bd7 100644 --- a/specs/platform/cluster-cpu.spec.md +++ b/specs/platform/cluster-cpu.spec.md @@ -23,7 +23,7 @@ The operational dashboard `cpu` widget and `system-summary` row already exist as - **Operational dashboard** (`web-console/operational-dashboard.spec.md`) owns the `cpu` widget, `UtilizationChart` presentation (OP-DASH-13), refresh policy (OP-DASH-09), and dashboard-operator access (OP-DASH-04). - **Cluster memory** (`platform/cluster-memory.spec.md`) follows the same Prometheus/BFF/adapter pattern for the `memory` widget. Cluster CPU reuses the same node-exporter scrape targets but exposes a separate BFF route and `cpu` metric mapping. -- **Registered users** (`platform/registered-users.spec.md`) sources HyperShell REST APIs instead of Prometheus. +- **Registered users** (`platform/registered-users.spec.md`) sources Prometheus via BFF `GET /api/metrics/registered-users`. - **Gateway metrics dashboard** (`platform/gateway-metrics-dashboard.spec.md`) uses a different BFF route (`GET /api/metrics/gateways`) for gateway phase counts and is unrelated to hub-cluster CPU. Pod capacity and provision-time metrics are out of scope for this spec (see Non-Goals). diff --git a/specs/platform/cluster-memory.spec.md b/specs/platform/cluster-memory.spec.md index c28c1cde8..a7158b95c 100644 --- a/specs/platform/cluster-memory.spec.md +++ b/specs/platform/cluster-memory.spec.md @@ -22,9 +22,9 @@ The operational dashboard `memory` widget and `system-summary` row already exist ### Relationship to other specifications - **Operational dashboard** (`web-console/operational-dashboard.spec.md`) owns the `memory` widget, `UtilizationChart` presentation (OP-DASH-13), refresh policy (OP-DASH-09), and dashboard-operator access (OP-DASH-04). -- **Registered users** (`platform/registered-users.spec.md`) follows the same dashboard-adapter pattern but sources HyperShell REST APIs. Cluster memory sources **Prometheus** through a BFF proxy because there is no durable HyperShell `Memory` resource in the API server database. -- **Gateway metrics dashboard** (`platform/gateway-metrics-dashboard.spec.md`) also uses a BFF Prometheus proxy (`GET /api/metrics/gateways`) for a different UI (`GatewayMetricsDashboard`). The operational dashboard does **not** reuse that route; it needs aggregate used/capacity bytes, not gateway phase counts. -- **Gateway list metrics** on `/dashboard` (`provisioned-gateways`, etc.) remain REST-driven and are unrelated. +- **Registered users** (`platform/registered-users.spec.md`) follows the same Prometheus-first dashboard pattern via BFF `GET /api/metrics/registered-users`. Cluster memory sources **Prometheus** through a BFF proxy because there is no durable HyperShell `Memory` resource in the API server database. +- **Gateway metrics dashboard** (`platform/gateway-metrics-dashboard.spec.md`) also uses a BFF Prometheus proxy (`GET /api/metrics/gateways`) for a different UI (`GatewayMetricsDashboard`). The operational dashboard reuses that route for gateway phase counts and adds separate routes for sandboxes, inventory, and registered users. +- **Gateway collection table** on `/gateways` remains REST-driven and is unrelated to cluster memory metrics. CPU, pod capacity, and node inventory are out of scope for this spec (see Non-Goals). diff --git a/specs/platform/cluster-nodes.spec.md b/specs/platform/cluster-nodes.spec.md index aead50f69..2016ada7b 100644 --- a/specs/platform/cluster-nodes.spec.md +++ b/specs/platform/cluster-nodes.spec.md @@ -25,7 +25,7 @@ The operational dashboard `system-summary` row and `nodes` widget render node in - **Cluster memory**, **cluster CPU**, and **cluster pods** (`platform/cluster-memory.spec.md`, `platform/cluster-cpu.spec.md`, `platform/cluster-pods.spec.md`) follow the same Prometheus/BFF/adapter pattern but expose **utilization** metrics with `unit` and `total`. Cluster nodes exposes **inventory + health buckets** (`value` + `status`), not utilization. - **Provisioned gateways** (`web-console/operational-dashboard.spec.md` OP-DASH-07) is the presentation reference: total in `value`, per-bucket counts in `status`, exception icons in summary rows when `failed` or `degraded` counts are non-zero. - **Cluster pods** (`platform/cluster-pods.spec.md`) reuses the same kube-state-metrics scrape target deployed for pod capacity series. -- **Registered users** (`platform/registered-users.spec.md`) sources HyperShell REST APIs instead of Prometheus. +- **Registered users** (`platform/registered-users.spec.md`) sources Prometheus via BFF `GET /api/metrics/registered-users`. - **Gateway metrics dashboard** (`platform/gateway-metrics-dashboard.spec.md`) is unrelated to hub-cluster node inventory. Provision-time metrics and per-node CPU/memory breakdown are out of scope for this spec (see Non-Goals). diff --git a/specs/platform/cluster-pods.spec.md b/specs/platform/cluster-pods.spec.md index 61c4a64e2..6f69f7b8c 100644 --- a/specs/platform/cluster-pods.spec.md +++ b/specs/platform/cluster-pods.spec.md @@ -24,7 +24,7 @@ The operational dashboard `pods` widget and `system-summary` row already exist a - **Operational dashboard** (`web-console/operational-dashboard.spec.md`) owns the `pods` widget, `UtilizationChart` presentation (OP-DASH-13), refresh policy (OP-DASH-09), and dashboard-operator access (OP-DASH-04). - **Cluster memory** and **cluster CPU** (`platform/cluster-memory.spec.md`, `platform/cluster-cpu.spec.md`) follow the same Prometheus/BFF/adapter pattern for utilization widgets. Cluster pods uses **kube-state-metrics** (not node-exporter) and exposes a separate BFF route and `pods` metric mapping. - **Provisioned sandboxes** (`platform/openshell-gateway-sandbox-count.spec.md`) counts gateway sandbox pods for product telemetry. Cluster pods counts **all** hub-cluster pods and is unrelated to sandbox lifecycle. -- **Registered users** (`platform/registered-users.spec.md`) sources HyperShell REST APIs instead of Prometheus. +- **Registered users** (`platform/registered-users.spec.md`) sources Prometheus via BFF `GET /api/metrics/registered-users`. - **Gateway metrics dashboard** (`platform/gateway-metrics-dashboard.spec.md`) uses a different BFF route (`GET /api/metrics/gateways`) and is unrelated to hub-cluster pod capacity. Node inventory and provision-time metrics are out of scope for this spec (see Non-Goals). diff --git a/specs/platform/gateway-metrics-dashboard.spec.md b/specs/platform/gateway-metrics-dashboard.spec.md index 24ed96de4..d4abba482 100644 --- a/specs/platform/gateway-metrics-dashboard.spec.md +++ b/specs/platform/gateway-metrics-dashboard.spec.md @@ -11,9 +11,9 @@ This specification covers the metrics pipeline end to end: the API server collec ### Relationship to the operational dashboard -The widgetized **operational dashboard** at `/dashboard` is specified separately in `web-console/operational-dashboard.spec.md`. That surface loads RBAC-scoped gateway counts from the HyperShell REST API and uses display-status buckets (`healthy`, `provisioning`, `degraded`, `failed`). It does **not** consume `GET /api/metrics/gateways` or `hypershell_gateways_total`. +The widgetized **operational dashboard** at `/dashboard` is specified separately in `web-console/operational-dashboard.spec.md`. That surface loads fleet-wide gateway phase counts from the same BFF route (`GET /api/metrics/gateways`, Prometheus `hypershell_gateways_total`) and maps phases to display-status buckets (`healthy`, `provisioning`, `degraded`, `failed`). -`GatewayMetricsDashboard` remains the canonical component for Prometheus-sourced phase counts. Hosts MAY embed it on any route; the `/dashboard` route is owned by the operational dashboard spec and renders `OperationalDashboardPage` instead. The web console exposes Prometheus phase counts at `/metrics`. +`GatewayMetricsDashboard` remains the canonical embeddable component for raw phase counts. The web console exposes Prometheus phase counts at `/metrics`. The operational dashboard and `GatewayMetricsDashboard` share the Prometheus pipeline but present different widgets and layouts. ## Requirements @@ -136,7 +136,7 @@ The `/api/metrics/gateways` route SHALL be exempt from the general `/api/*` prox When OIDC is enabled, the route SHALL require **dashboard-operator authorization** matching `web-console/operational-dashboard.spec.md` OP-DASH-04 (`hypershell-admins` or `platform:admin`). Authenticated callers without a dashboard-admin role SHALL receive HTTP `403`. Unauthenticated callers SHALL receive HTTP `401` or the standard BFF re-authentication response. When OIDC is disabled, no session or role is required. -Fleet-wide phase counts from Prometheus are intentionally **not** filtered by per-gateway RoleBindings; this route is restricted to dashboard administrators who are authorized to view platform-wide operational data. Per-user gateway visibility for the operational dashboard gateway-status widget remains on the RBAC-scoped HyperShell REST list API (`operational-dashboard.spec.md` OP-DASH-06). +Fleet-wide phase counts from Prometheus are intentionally **not** filtered by per-gateway RoleBindings; this route is restricted to dashboard administrators who are authorized to view platform-wide operational data. The operational dashboard gateway-status widget consumes this same route (`operational-dashboard.spec.md` OP-DASH-23). Per-user gateway visibility for the gateway collection table remains on the RBAC-scoped HyperShell REST list API (`web-console/architecture.spec.md` WEB-DATA-01). #### Scenario: Successful metrics fetch diff --git a/specs/platform/gateway-provision-time.spec.md b/specs/platform/gateway-provision-time.spec.md index 4dfb6ce8a..612155f39 100644 --- a/specs/platform/gateway-provision-time.spec.md +++ b/specs/platform/gateway-provision-time.spec.md @@ -212,13 +212,13 @@ When `provisionDuration` is absent but `value` and `unit` are present, the UI MA Provision-time collection SHALL NOT depend on the paginated gateway list (`GET /api/hypershell/v1/gateways`). -Gateway list pagination failure SHALL omit `provisioned-gateways` and `provisioned-sandboxes` only. It SHALL NOT omit `provision-time` when the BFF provision-duration route succeeds. +Gateway metrics source failure (BFF `GET /api/metrics/gateways` or `GET /api/metrics/gateway-sandboxes`) SHALL omit `provisioned-gateways` and `provisioned-sandboxes` only. It SHALL NOT omit `provision-time` when the BFF provision-duration route succeeds. -Conversely, provision-duration BFF failure SHALL omit only `provision-time`. It SHALL NOT affect gateway-list-derived metrics (OP-DASH-19). +Conversely, provision-duration BFF failure SHALL omit only `provision-time`. It SHALL NOT affect other gateway-metrics-derived counts (OP-DASH-19). -#### Scenario: Gateway list down does not hide provision time +#### Scenario: Gateway metrics down does not hide provision time -- GIVEN the gateway list request fails +- GIVEN the `gateway-metrics` source fails (for example, `GET /api/metrics/gateways` returns HTTP `502`) - AND `GET /api/metrics/gateway-provision-duration` succeeds - WHEN the operator opens `/dashboard` - THEN `provision-time` SHALL appear in the adapter response with mean, P50, and P95 @@ -237,7 +237,7 @@ When provision-duration collection fails (BFF `502`, zero observations, non-fini - GIVEN every other metric source succeeds - AND `gateway_provision_duration_seconds_count` is zero - WHEN the operator opens `/dashboard` -- THEN cluster and gateway-list metrics SHALL still load +- THEN gateway-metrics and cluster metrics SHALL still load - AND the provision-time summary rows SHALL render the localized metric-unavailable state - AND the dashboard SHALL NOT enter the total load-error state @@ -250,7 +250,7 @@ The web console SHALL include unit tests for: - BFF route PromQL mapping and JSON response formatting (including `observation_count`) - BFF `502` on zero count, Prometheus errors, and non-finite quantiles - Adapter mapping from BFF JSON to `provision-time` with `provisionDuration.mean`, `.p50`, and `.p95` -- Independent source behavior: gateway list failure does not block provision time and vice versa (GPT-06) +- Independent source behavior: gateway-metrics source failure does not block provision time and vice versa (GPT-06) The operational dashboard package SHALL include unit tests or Storybook fixtures for the three-row system-summary presentation when `provisionDuration` is present. diff --git a/specs/platform/openshell-gateway-sandbox-count.spec.md b/specs/platform/openshell-gateway-sandbox-count.spec.md index 398bfec7f..635658a5c 100644 --- a/specs/platform/openshell-gateway-sandbox-count.spec.md +++ b/specs/platform/openshell-gateway-sandbox-count.spec.md @@ -186,6 +186,16 @@ It surfaces load to operators; it never blocks an action. - THEN the deletion SHALL proceed regardless of the reported count, which is used only to warn the operator beforehand +### Requirement: Operational Dashboard Fleet Aggregate + +The API server SHALL expose a Prometheus gauge `hypershell_gateways_active_sandboxes_total` that sums `active_sandbox_count` across all gateways on each scrape. The operational dashboard loads this fleet-wide total through BFF `GET /api/metrics/gateway-sandboxes` (`web-console/operational-dashboard.spec.md` OP-DASH-06). This aggregate is dashboard-operator scoped and is independent of per-gateway RBAC visibility on the gateway collection table. + +#### Scenario: Fleet sandbox gauge reflects database sum + +- GIVEN three gateways report `active_sandbox_count` values of `2`, `0`, and `3` +- WHEN Prometheus scrapes the API server `/metrics` endpoint +- THEN `hypershell_gateways_active_sandboxes_total` SHALL report `5` + ## Design Decisions | Decision | Rationale | diff --git a/specs/platform/platform-inventory.spec.md b/specs/platform/platform-inventory.spec.md index a6993c25e..818d86ca2 100644 --- a/specs/platform/platform-inventory.spec.md +++ b/specs/platform/platform-inventory.spec.md @@ -9,12 +9,14 @@ Expose **platform inventory counts** - totals and breakdowns for HyperShell infrastructure resources registered in the API server - on the operational dashboard so administrators can assess platform footprint at a glance. -Version 1 sources inventory from existing paginated List APIs: +Version 1 sources operational dashboard inventory from Prometheus gauges emitted by API server inventory collectors and surfaced through BFF `GET /api/metrics/platform-inventory`. The underlying data still lives in the API server database; collectors query it on each Prometheus scrape. -- `GET /api/hypershell/v1/managed_clusters` +HyperShell REST List APIs remain available for other consumers: + +- `GET /api/hypershell/v1/managed_clusters` (gateway placement selection, bounded search) - `GET /api/hypershell/v1/managed_databases` -This specification defines aggregation rules, dashboard-operator authorization, operational dashboard metric IDs, and the **Inventory summary** presentation. It does **not** add usage trend graphs or new inventory APIs. +This specification defines aggregation rules, dashboard-operator authorization, operational dashboard metric IDs, Prometheus collectors, the BFF proxy route, and the **Inventory summary** presentation. It does **not** add usage trend graphs. ### Fleet resource exclusion @@ -24,12 +26,12 @@ HyperShell removed the top-level Fleet resource and `fleet_id` scoping from all | Concern | Platform inventory (this spec) | Operational dashboard | | --- | --- | --- | -| Data source | HyperShell REST list APIs (RBAC-filtered) | Widget layout, refresh, access control | -| Managed cluster placement UI | Gateway provisioning uses bounded list search (`web-console/architecture.spec.md`) | Inventory adapter paginates the full collection for aggregates | +| Data source | Prometheus inventory gauges via BFF `GET /api/metrics/platform-inventory` | Widget layout, refresh, access control | +| Managed cluster placement UI | Gateway provisioning uses bounded REST list search (`web-console/architecture.spec.md`) | Inventory adapter loads Prometheus-backed aggregates | | Hub cluster nodes | Unrelated (Prometheus kube-state-metrics; `platform/cluster-nodes.spec.md`) | `nodes` widget counts hub-cluster Kubernetes nodes, not ManagedCluster registrations | -| Registered users | Same list-total pattern (`platform/registered-users.spec.md`) | Both appear on the operational overview | +| Registered users | Same Prometheus-first dashboard pattern (`platform/registered-users.spec.md`) | Both appear on the operational overview | -Prometheus gateway metrics (`platform/gateway-metrics-dashboard.spec.md`) are unrelated. +Prometheus gateway metrics (`platform/gateway-metrics-dashboard.spec.md`) cover gateway phase counts, not managed cluster or database inventory. ## Requirements @@ -55,36 +57,38 @@ Historical trend series (`OperationalMetric.trend`) SHALL NOT be loaded in versi --- -### Requirement: PI-02 -- Paginated List Aggregation +### Requirement: PI-02 -- Prometheus Inventory Metrics + +The operational dashboard adapter SHALL load managed cluster and managed database inventory from BFF `GET /api/metrics/platform-inventory`, which queries Prometheus gauges emitted by API server inventory collectors on each scrape. -The web-console dashboard adapter SHALL load managed cluster and managed database inventory by paginating each List API through the browser TypeScript SDK with page size `100`, ordered by `name asc`, until all pages are retrieved. +Managed cluster Prometheus series: -The adapter SHALL validate each list response for internal consistency (requested page number, `total`, and `items` length). An inconsistent response SHALL fail only the `platform-inventory` metric source; other metric sources SHALL still be attempted (`web-console/operational-dashboard.spec.md` OP-DASH-19). +| Metric | Meaning | +| --- | --- | +| `hypershell_managed_clusters_total` | Total registered managed clusters | +| `hypershell_managed_clusters_created_last_30_days_total` | Clusters created in the last 30 × 24 hours (UTC) | +| `hypershell_managed_clusters_inventory_total{status, provider, region}` | Cluster counts by inventory dimensions | -For each resource kind the adapter SHALL compute: +Managed database Prometheus series: -| Aggregate | Rule | +| Metric | Meaning | | --- | --- | -| `total` | List `total` after the final page (MUST match the number of aggregated items) | -| `status` buckets | Group by each item's `status` field; omitted or null `status` SHALL bucket as `unknown` | -| `provider` buckets (clusters only) | Group by each item's `provider` field; omitted or null `provider` SHALL bucket as `unknown` | -| `region` buckets (clusters only) | Group by each item's `region` field; omitted or null `region` SHALL bucket as `unknown` | -| `created_last_30_days` | Count items whose `created_at` is greater than or equal to the start of the 30-day lookback window | +| `hypershell_managed_databases_total` | Total registered managed databases | +| `hypershell_managed_databases_inventory_total{status}` | Database counts by status | -The 30-day lookback window SHALL be **30 × 24 hours** ending at adapter evaluation time (UTC). Items with missing or unparsable `created_at` SHALL be excluded from the recent count but SHALL still contribute to `total` and breakdown buckets. +The adapter SHALL map the BFF JSON response into `managed-clusters` and `managed-databases` operational metrics per PI-04 and PI-05. A non-success BFF response SHALL fail only the `platform-inventory` metric source (`web-console/operational-dashboard.spec.md` OP-DASH-19). -The adapter SHALL NOT issue per-resource Get requests. Inventory aggregation SHALL use at most one paginated List sequence per resource kind per metrics refresh. +The adapter SHALL NOT paginate HyperShell REST List APIs for dashboard inventory aggregates. REST List APIs remain authoritative for gateway placement selection and other collection workflows. -#### Scenario: Multiple pages are aggregated +#### Scenario: Platform inventory BFF populates managed cluster total -- GIVEN the caller can list 150 managed clusters +- GIVEN `GET /api/metrics/platform-inventory` returns `managed_clusters.total: 150` - WHEN `getOperationalMetrics` runs -- THEN the adapter SHALL issue two paginated managed cluster list requests -- AND the `managed-clusters` metric `value` SHALL be `"150"` +- THEN the `managed-clusters` metric `value` SHALL be `"150"` -#### Scenario: Inconsistent pagination omits inventory metrics +#### Scenario: Platform inventory BFF failure omits inventory metrics -- GIVEN a managed database list response reports `page: 2` when page `1` was requested +- GIVEN `GET /api/metrics/platform-inventory` fails - AND at least one other metric source succeeds - WHEN the adapter processes the response - THEN the `platform-inventory` source SHALL be treated as failed @@ -94,6 +98,28 @@ The adapter SHALL NOT issue per-resource Get requests. Inventory aggregation SHA --- +### Requirement: PI-10 -- API Server Inventory Collectors and BFF Route + +The API server SHALL register Prometheus collectors for managed cluster and managed database inventory. Each collector SHALL query the database once per scrape via an inventory snapshot DAO method and emit gauges matching PI-02. When the database query fails, the collector SHALL emit `prometheus.NewInvalidMetric` so the scrape registers as failed. + +The web-console BFF SHALL expose `GET /api/metrics/platform-inventory` as a same-origin proxy route that queries Prometheus instant vectors and scalars and returns structured JSON for managed clusters and managed databases. The route SHALL use the same `PROMETHEUS_URL` and `PROMETHEUS_QUERY_TIMEOUT_MS` configuration as other `/api/metrics/*` routes. + +When OIDC is enabled, the route SHALL require dashboard-operator authorization matching `web-console/operational-dashboard.spec.md` OP-DASH-04. When Prometheus is unreachable or returns a non-success response, the BFF SHALL respond with HTTP `502` and `{ "error": "Metrics unavailable", "statusCode": 502 }`. + +#### Scenario: Collector emits inventory gauges on scrape + +- GIVEN 42 managed clusters exist with known status, provider, and region buckets +- WHEN Prometheus scrapes the API server `/metrics` endpoint +- THEN samples for `hypershell_managed_clusters_total`, `hypershell_managed_clusters_created_last_30_days_total`, and `hypershell_managed_clusters_inventory_total` SHALL be present + +#### Scenario: BFF maps Prometheus inventory into JSON + +- GIVEN Prometheus returns current inventory gauge values +- WHEN an authorized caller sends `GET /api/metrics/platform-inventory` +- THEN the BFF SHALL respond with HTTP `200` and JSON containing `managed_clusters` and `managed_databases` totals and breakdown maps + +--- + ### Requirement: PI-03 -- Dashboard-Operator Authorization Managed cluster and managed database List endpoints SHALL be readable by **dashboard operators**, matching the operational dashboard audience (`web-console/operational-dashboard.spec.md` OP-DASH-04) and registered user inventory (`platform/registered-users.spec.md` RU-03): @@ -230,17 +256,17 @@ A dedicated **Platform inventory** dashboard route (`/dashboard/inventory`) SHAL Platform inventory metrics SHALL load through the existing operational dashboard metrics query (`useGetMetricsData`) and SHALL inherit its refresh policy (`operationalDashboardRefreshMilliseconds`, currently 15 minutes) and manual refresh behavior (`web-console/operational-dashboard.spec.md` OP-DASH-09). -A failed managed cluster or managed database List request SHALL fail only the `platform-inventory` metric source (`managed-clusters`, `managed-databases`). The adapter SHALL NOT synthesize zero, empty, or placeholder values for failed inventory metrics. When at least one other metric source succeeds, the dashboard SHALL render available metrics and show inventory widgets in the localized metric-unavailable state (`web-console/operational-dashboard.spec.md` OP-DASH-08, OP-DASH-19). +A failed `GET /api/metrics/platform-inventory` request SHALL fail only the `platform-inventory` metric source (`managed-clusters`, `managed-databases`). The adapter SHALL NOT synthesize zero, empty, or placeholder values for failed inventory metrics. When at least one other metric source succeeds, the dashboard SHALL render available metrics and show inventory widgets in the localized metric-unavailable state (`web-console/operational-dashboard.spec.md` OP-DASH-08, OP-DASH-19). `getOperationalMetrics` SHALL throw only when every metric source fails or when the request is aborted. -`AbortSignal` cancellation SHALL propagate to in-flight list requests. +`AbortSignal` cancellation SHALL propagate to in-flight BFF requests. -#### Scenario: Unauthorized inventory list omits inventory metrics +#### Scenario: Unauthorized platform-inventory BFF call omits inventory metrics -- GIVEN the signed-in user lacks dashboard-operator API authorization for managed clusters +- GIVEN the signed-in user lacks dashboard-operator BFF authorization - AND at least one other metric source succeeds -- WHEN the host adapter calls `GET /api/hypershell/v1/managed_clusters` +- WHEN the host adapter calls `GET /api/metrics/platform-inventory` - THEN the `platform-inventory` source SHALL be treated as failed - AND inventory widgets SHALL render the localized metric-unavailable state - AND a warning `Alert` SHALL explain that some metrics could not be loaded @@ -249,22 +275,22 @@ A failed managed cluster or managed database List request SHALL fail only the `p ### Requirement: PI-09 -- Documentation and Verification -`packages/operational-dashboard-ui/DATA_SOURCES.md` SHALL document `managed-clusters` and `managed-databases` metric sources, pagination, aggregation rules, and the inventory summary widget. +`packages/operational-dashboard-ui/DATA_SOURCES.md` SHALL document `managed-clusters` and `managed-databases` metric sources, the BFF `platform-inventory` route, aggregation rules, and the inventory summary widget. The web console SHALL include unit tests for the dashboard adapter that cover: -- Full pagination aggregation +- BFF `platform-inventory` response mapping into operational metrics - `unknown` bucketing for omitted `status`, `provider`, and `region` -- `created_last_30_days` counting with a fixed clock or injected timestamps +- `created_last_30_days` passthrough from Prometheus gauges - Provider and region bucket mapping into `inventoryProviders` and `inventoryRegions` - Metric ID and field mapping into `OperationalDashboardMetrics` -The API server SHALL include RBAC tests for dashboard-operator List access to `managed_clusters` and `managed_databases`. +The API server SHALL include RBAC tests for dashboard-operator List access to `managed_clusters` and `managed_databases`, and unit tests for inventory snapshot DAO methods used by the Prometheus collectors. The operational dashboard package SHALL extend `mockOperationalDashboardMetrics` with representative inventory fields and add Storybook coverage for the inventory summary widget. #### Scenario: CI exercises adapter mapping -- GIVEN a managed cluster list fixture with two pages and mixed field presence +- GIVEN a `platform-inventory` BFF fixture with mixed inventory buckets - WHEN dashboard adapter unit tests run - THEN they SHALL assert the stringified total, `inventoryStatus` buckets, `createdLast30Days`, `inventoryProviders`, and `inventoryRegions` mapping diff --git a/specs/platform/registered-users.spec.md b/specs/platform/registered-users.spec.md index d8ccae27a..fc58fb881 100644 --- a/specs/platform/registered-users.spec.md +++ b/specs/platform/registered-users.spec.md @@ -13,9 +13,9 @@ User creation remains middleware-driven (see `security/rbac-enforcement.spec.md` ### Relationship to the operational dashboard -The operational dashboard widget currently labeled "Active users" (`active-users`) is a placeholder. This spec introduces the metric ID `registered-users`, connects it to the users List API, and renames user-facing copy to **Registered users** so the UI matches the data semantics. +The operational dashboard widget currently labeled "Active users" (`active-users`) is a placeholder. This spec introduces the metric ID `registered-users`, connects it to Prometheus `hypershell_users_registered_total` via BFF `GET /api/metrics/registered-users`, and renames user-facing copy to **Registered users** so the UI matches the data semantics. -Prometheus gateway metrics (`platform/gateway-metrics-dashboard.spec.md`) are unrelated. +The users List API remains available for future collection workflows but is not the preferred source for the operational dashboard total count. ## Requirements @@ -125,9 +125,9 @@ The List response SHALL include accurate `page`, `size`, `total`, and `items` fi ### Requirement: RU-05 -- Operational Dashboard Metric -The operational dashboard host adapter SHALL populate an `OperationalMetric` with `id: "registered-users"` and `value` set to the decimal string of the user List `total`. +The operational dashboard host adapter SHALL populate an `OperationalMetric` with `id: "registered-users"` and `value` set to the decimal string of the registered user total. -The adapter SHALL obtain `total` from the HyperShell REST API through the browser TypeScript SDK. It SHOULD use a single List request with `page=1` and `size=1` to minimize payload size; it SHALL NOT paginate through all user records when only the count is required. +The adapter SHALL obtain the total from BFF `GET /api/metrics/registered-users`, which queries Prometheus for `hypershell_users_registered_total` emitted by the API server users metrics collector on each scrape. The adapter SHALL NOT paginate the users List API when only the count is required. The metric SHALL NOT include `trend`, `status`, `unit`, or `total` fields in version 1. @@ -141,17 +141,37 @@ The operational dashboard package SHALL rename the widget and summary labels fro - THEN the `registered-users` metric SHALL have `value: "42"` - AND the usage summary row SHALL display `42` under **Registered users** -#### Scenario: Unauthorized adapter call omits registered-users metric +#### Scenario: Unauthorized BFF call omits registered-users metric -- GIVEN the signed-in user lacks dashboard-operator API authorization +- GIVEN the signed-in user lacks dashboard-operator BFF authorization - AND at least one other metric source succeeds -- WHEN the host adapter calls `GET /api/hypershell/v1/users` +- WHEN the host adapter calls `GET /api/metrics/registered-users` - THEN the `registered-users` metric SHALL be omitted from the adapter response - AND the dashboard SHALL show its localized partial-load warning (OP-DASH-09) - AND the registered-users widget and usage-summary row SHALL render the localized metric-unavailable state (not a silent zero count) --- +### Requirement: RU-09 -- Registered Users Prometheus Collector and BFF Route + +The API server SHALL register a Prometheus collector that emits `hypershell_users_registered_total`, querying `CountRegistered` from the users DAO on each scrape. When the database query fails, the collector SHALL emit `prometheus.NewInvalidMetric`. + +The web-console BFF SHALL expose `GET /api/metrics/registered-users` as a same-origin proxy route that queries Prometheus and returns `{ "total_registered": N }`. Dashboard-operator authorization SHALL match `web-console/operational-dashboard.spec.md` OP-DASH-04. When Prometheus is unreachable, the BFF SHALL respond with HTTP `502`. + +#### Scenario: Collector emits registered user gauge on scrape + +- GIVEN 42 registered users exist in the database +- WHEN Prometheus scrapes the API server `/metrics` endpoint +- THEN a sample for `hypershell_users_registered_total` with value `42` SHALL be present + +#### Scenario: BFF returns registered user total + +- GIVEN Prometheus returns `hypershell_users_registered_total` with value `42` +- WHEN an authorized caller sends `GET /api/metrics/registered-users` +- THEN the BFF SHALL respond with HTTP `200` and `{ "total_registered": 42 }` + +--- + ### Requirement: RU-06 -- UI Presentation The `registered-users` widget SHALL render through the existing `MetricCard` presentation (large numeric heading, localized title). @@ -173,7 +193,7 @@ All user-visible strings SHALL use `defineMessages` in `operational-dashboard-ui Registered user counts SHALL load through the existing operational dashboard metrics query (`useGetMetricsData`) and SHALL inherit its refresh policy (`operationalDashboardRefreshMilliseconds`, currently 15 minutes) and manual refresh behavior defined in `web-console/operational-dashboard.spec.md` OP-DASH-09. -A failed users List request SHALL fail only the registered-users metric source (OP-DASH-19); the dashboard SHALL NOT display `0` as a fallback count. +A failed `GET /api/metrics/registered-users` request SHALL fail only the registered-users metric source (OP-DASH-19); the dashboard SHALL NOT display `0` as a fallback count. #### Scenario: Refresh updates the displayed total @@ -193,7 +213,7 @@ The API server SHALL include integration tests for: - Opaque 404 on unauthorized singleton Get - Accurate `total` with `size=1` -The web console SHALL include unit tests for the dashboard adapter mapping `UserList.total` into `registered-users`. +The web console SHALL include unit tests for the dashboard adapter mapping BFF `GET /api/metrics/registered-users` response into `registered-users`. The operational dashboard package SHALL update Storybook fixtures and `mockOperationalDashboardMetrics` to use `registered-users` instead of `active-users`. diff --git a/specs/web-console/operational-dashboard.spec.md b/specs/web-console/operational-dashboard.spec.md index db6269b17..798f08e41 100644 --- a/specs/web-console/operational-dashboard.spec.md +++ b/specs/web-console/operational-dashboard.spec.md @@ -1,24 +1,26 @@ # Operational Dashboard **Status:** Active -**Applies to:** `packages/operational-dashboard-ui`, `components/web-console` SPA and BFF, `packages/gateway-management-ui` (display-status aggregation), `components/sdk-typescript` (gateway list client) +**Applies to:** `packages/operational-dashboard-ui`, `components/web-console` SPA and BFF, `packages/gateway-management-ui` (display-status aggregation) ## Purpose Provide a widgetized operational dashboard in the HyperShell web console where administrators can assess fleet health at a glance. The dashboard composes summary and detail widgets in a customizable grid layout. Live data is loaded through a narrow application port (`DashboardControlPlane`) implemented by the web-console host; widgets without a connected source remain on the page and render a localized unavailable state rather than being hidden. -This specification covers the reusable `operational-dashboard-ui` package, the host adapter that aggregates gateway list data, admin-only access controls, SPA and BFF route surfaces, layout persistence, and the widget catalog. Platform inventory metrics (managed clusters and managed databases) are defined in `platform/platform-inventory.spec.md`. It does **not** cover the Prometheus metrics pipeline (`hypershell_gateways_total`, BFF `GET /api/metrics/gateways`, or `GatewayMetricsDashboard`); those are defined in `platform/gateway-metrics-dashboard.spec.md`. +This specification covers the reusable `operational-dashboard-ui` package, the host adapter that loads operational metrics from BFF Prometheus proxy routes, admin-only access controls, SPA and BFF route surfaces, layout persistence, and the widget catalog. Platform inventory metrics (managed clusters and managed databases) are defined in `platform/platform-inventory.spec.md`. Prometheus gateway phase counts (`hypershell_gateways_total`, BFF `GET /api/metrics/gateways`) are defined in `platform/gateway-metrics-dashboard.spec.md` and consumed by this dashboard for `provisioned-gateways` and `gateway-status`. The embeddable `GatewayMetricsDashboard` component remains a separate surface. + +**Data-source direction:** Aggregate operational dashboard widgets SHALL prefer Prometheus-backed BFF routes over direct HyperShell REST list pagination. The API server exposes fleet-wide gauges from database state on its `/metrics` scrape; the BFF queries Prometheus and returns JSON to the browser adapter. HyperShell REST List APIs remain authoritative for resource collection pages (for example, the gateway table) but are not the preferred source for dashboard totals. ### Relationship to gateway metrics | Concern | Operational dashboard (this spec) | Gateway metrics dashboard | | --- | --- | --- | | Primary route | `/dashboard` and `dashboard.*` host root (`/`) | `GatewayMetricsDashboard` component (embeddable) | -| Gateway counts source | Paginated HyperShell REST `GET /api/hypershell/v1/gateways` | Prometheus `hypershell_gateways_total` via BFF proxy | -| Scope | RBAC-filtered: counts reflect gateways visible to the signed-in caller | Fleet-wide database aggregate | -| Status model | Display buckets: `healthy`, `provisioning`, `degraded`, `failed` | Lifecycle phases: `Running`, `Provisioning`, `Degraded`, `Failed` | +| Gateway counts source | BFF `GET /api/metrics/gateways` (Prometheus `hypershell_gateways_total`) | Same BFF route | +| Scope | Fleet-wide database aggregate (dashboard-operator access only) | Fleet-wide database aggregate (dashboard-operator access only) | +| Status model | Display buckets: `healthy`, `provisioning`, `degraded`, `failed` (mapped from phase labels) | Lifecycle phases: `Pending`, `Provisioning`, `Running`, `Degraded`, `Failed` | -The two surfaces MAY coexist. The operational dashboard SHALL NOT be required to consume the Prometheus BFF route. +The two surfaces MAY coexist. They share the same Prometheus-backed gateway count route but present different widgets and layouts. ## Requirements @@ -90,7 +92,7 @@ The SPA route modules for `/dashboard` and the dashboard-host root (`/`) SHALL w When OIDC is enabled, the BFF SHALL enforce the same role requirement for browser navigations to `/dashboard` and for `/` on hosts whose hostname starts with `dashboard.`. Non-admin users SHALL be redirected away (to `/` on the console host, or to the console host when the request arrived on a dashboard subdomain). -When OIDC is enabled, the BFF SHALL enforce the same dashboard-admin role requirement on every `GET /api/metrics/*` route consumed by the operational dashboard host adapter (`cluster-memory`, `cluster-cpu`, `cluster-pods`, `cluster-nodes`, and `gateway-provision-duration` per OP-DASH-08). Authenticated callers without a dashboard-admin role SHALL receive HTTP `403`. Unauthenticated callers SHALL receive HTTP `401` or the standard BFF re-authentication response. When OIDC is disabled (no-auth dev mode), these routes SHALL remain open to unauthenticated callers, matching page behavior. +When OIDC is enabled, the BFF SHALL enforce the same dashboard-admin role requirement on every `GET /api/metrics/*` route consumed by the operational dashboard host adapter (`cluster-memory`, `cluster-cpu`, `cluster-pods`, `cluster-nodes`, `gateways`, `gateway-sandboxes`, `gateway-provision-duration`, `platform-inventory`, and `registered-users` per OP-DASH-08). Authenticated callers without a dashboard-admin role SHALL receive HTTP `403`. Unauthenticated callers SHALL receive HTTP `401` or the standard BFF re-authentication response. When OIDC is disabled (no-auth dev mode), these routes SHALL remain open to unauthenticated callers, matching page behavior. #### Scenario: Non-admin is turned away from /dashboard @@ -144,63 +146,84 @@ When the browser hostname is `dashboard.hypershell.localhost`, the SPA root rout --- -### Requirement: OP-DASH-06 -- Gateway List Metrics Adapter +### Requirement: OP-DASH-06 -- Gateway Sandbox Metrics Adapter -The host `DashboardControlPlane` adapter SHALL load operational metrics by paginating `GET /api/hypershell/v1/gateways` through the browser TypeScript SDK with page size `100`, ordered by `name asc`, until all pages are retrieved. +The host `DashboardControlPlane` adapter SHALL load `provisioned-sandboxes` from BFF `GET /api/metrics/gateway-sandboxes` in the `gateway-metrics` metric source (OP-DASH-23). -The adapter SHALL validate each list response for internal consistency (page number, total count, and item count). An inconsistent response SHALL fail only the gateway-list metric source; other metric sources SHALL still be attempted (OP-DASH-19). +The BFF route SHALL query Prometheus for `hypershell_gateways_active_sandboxes_total`, a fleet-wide gauge emitted by the API server metrics collector that sums `active_sandbox_count` across all gateways on each scrape (see `openshell-gateway-sandbox-count.spec.md` for field semantics). -The adapter SHALL return `OperationalDashboardMetrics` containing: +The adapter SHALL emit a `provisioned-sandboxes` metric whose `value` is the stringified `active_sandboxes` count from the BFF response. The value SHALL NOT be a non-finite number. -- `lastSuccessfulRefresh` set to the current time -- A `provisioned-gateways` metric (see OP-DASH-07) -- A `provisioned-sandboxes` metric whose `value` is the stringified sum of `active_sandbox_count` across all gateways in the aggregated list +A non-success BFF response for gateway sandboxes SHALL fail the entire `gateway-metrics` source (including `provisioned-gateways` and `provision-time`). Access control is enforced at the BFF route (dashboard-operator roles per OP-DASH-04). + +#### Scenario: Prometheus sandbox count populates provisioned-sandboxes + +- GIVEN `GET /api/metrics/gateway-sandboxes` returns `{ "active_sandboxes": 5 }` +- WHEN `getOperationalMetrics` runs +- THEN the adapter SHALL emit `provisioned-sandboxes` with `value: "5"` +- AND the dashboard SHALL NOT display `NaN` -When summing `active_sandbox_count`, the adapter SHALL treat an omitted or null field on a gateway as `0` (matching database `COALESCE` semantics in `openshell-gateway-sandbox-count.spec.md`). The sum SHALL NOT produce a non-finite numeric result. +#### Scenario: Gateway sandboxes failure omits gateway-metrics source -Gateway list results SHALL reflect the caller's RBAC visibility (the API applies gateway visibility filtering; the dashboard does not bypass it). +- GIVEN `GET /api/metrics/gateway-sandboxes` fails +- WHEN the adapter processes the `gateway-metrics` source +- THEN the `gateway-metrics` source SHALL be treated as failed +- AND `provisioned-sandboxes`, `provisioned-gateways`, and `provision-time` SHALL be omitted +- AND the dashboard SHALL NOT synthesize a zero sandbox count -#### Scenario: Omitted sandbox counts do not break the aggregate +--- -- GIVEN the aggregated list contains gateways where some omit `active_sandbox_count` and others report `2` and `3` -- WHEN the adapter computes `provisioned-sandboxes` -- THEN the metric `value` SHALL be `"5"` -- AND the dashboard SHALL NOT display `NaN` +### Requirement: OP-DASH-23 -- Gateway Prometheus Metrics Adapter -#### Scenario: Multiple pages are aggregated +The host `DashboardControlPlane` adapter SHALL load `provisioned-gateways` (see OP-DASH-07), `provisioned-sandboxes` (OP-DASH-06), and optionally `provision-time` from BFF Prometheus proxy routes in the `gateway-metrics` metric source: -- GIVEN the caller can see 150 gateways +- `GET /api/metrics/gateways` - fleet-wide phase counts from `hypershell_gateways_total` (see `platform/gateway-metrics-dashboard.spec.md` DASH-05) +- `GET /api/metrics/gateway-sandboxes` - fleet-wide active sandbox sum from `hypershell_gateways_active_sandboxes_total` (OP-DASH-06) +- `GET /api/metrics/gateway-provision-duration` - provision duration histogram (see `platform/gateway-provision-time.spec.md`) + +The adapter SHALL call `fetchGatewayMetrics` from `@openshift-online/hypershell-gateway-management-ui` for phase counts. A non-success BFF response for gateways or gateway sandboxes SHALL fail the entire `gateway-metrics` source. + +`provision-time` SHALL be appended to the same source only when the provision-duration BFF route succeeds. When the route fails or returns no qualifying samples, `provision-time` SHALL be omitted while `provisioned-gateways` and `provisioned-sandboxes` MAY still be emitted. + +Gateway phase counts SHALL be fleet-wide and SHALL NOT be filtered by per-gateway RoleBindings. Access control is enforced at the BFF route (dashboard-operator roles per OP-DASH-04). + +#### Scenario: Prometheus gateway counts populate provisioned-gateways + +- GIVEN `GET /api/metrics/gateways` returns `{ "counts": { "Running": 10, "Provisioning": 3, "Degraded": 1, "Failed": 4, "Pending": 2 } }` - WHEN `getOperationalMetrics` runs -- THEN the adapter SHALL issue two paginated list requests -- AND the `provisioned-gateways` metric `value` SHALL be `"150"` +- THEN the adapter SHALL emit `provisioned-gateways` with `value: "20"` +- AND `status` SHALL map phases to display buckets per OP-DASH-07 -#### Scenario: Inconsistent pagination omits gateway-derived metrics +#### Scenario: Prometheus gateway failure omits gateway-derived Prometheus metrics -- GIVEN a list response reports `page: 2` when page `1` was requested -- WHEN the adapter processes the response -- THEN the gateway list source SHALL be treated as failed -- AND `provisioned-gateways`, `provisioned-sandboxes`, and `provision-time` SHALL be omitted from the adapter response -- AND the dashboard SHALL NOT synthesize zero counts for those metrics +- GIVEN `GET /api/metrics/gateways` fails +- WHEN the adapter processes the `gateway-metrics` source +- THEN the `gateway-metrics` source SHALL be treated as failed +- AND `provisioned-gateways`, `provisioned-sandboxes`, and `provision-time` SHALL be omitted --- ### Requirement: OP-DASH-07 -- Gateway Display Status Aggregation -Gateway status widgets SHALL use the same display-status presentation rules as the gateway list. The host adapter SHALL pass each gateway's `phase` and `status` to `aggregateGatewayDisplayStatusCounts` from `@openshift-online/hypershell-gateway-management-ui`. +Gateway status widgets SHALL present display buckets (`healthy`, `provisioning`, `degraded`, `failed`) aligned with the gateway list vocabulary. The host adapter SHALL map Prometheus phase counts to display buckets using `gatewayPhaseCountsToDisplayStatusCounts` from `@openshift-online/hypershell-gateway-management-ui`. + +Phase-to-bucket mapping SHALL treat `Pending` and `Provisioning` as `provisioning`, `Running` as `healthy`, `Degraded` as `degraded`, and `Failed` as `failed`. The `provisioned-gateways` metric SHALL include: -- `value` - total gateway count as a decimal string +- `value` - total gateway count as a decimal string (sum of all phase counts) - `status` - counts for `healthy`, `provisioning`, `degraded`, and `failed` display buckets -Display buckets SHALL NOT be confused with raw lifecycle `phase` values. Mapping from phase/status to display buckets SHALL remain owned by the gateway-management-ui package so the dashboard and gateway list stay aligned. +Display buckets SHALL NOT be confused with raw lifecycle `phase` values. Mapping logic SHALL remain owned by the gateway-management-ui package. -#### Scenario: Gateway status widget reflects list presentation +**Trade-off:** Prometheus exposes `phase` labels only. The gateway list combines `phase` and `status` (for example, `Running` with an unhealthy status resolves to `degraded`). Phase-only mapping MAY under-count `degraded` when phase is still `Running`. Finer-grained status alignment would require a Prometheus metric with both dimensions or a REST list aggregate. -- GIVEN the aggregated list contains gateways whose resolved display statuses are 5 healthy, 2 provisioning, 1 degraded, and 0 failed +#### Scenario: Gateway status widget reflects phase-mapped buckets + +- GIVEN Prometheus returns phase counts equivalent to 10 healthy, 5 provisioning, 1 degraded, and 4 failed after mapping - WHEN the gateway status widget renders -- THEN the donut chart SHALL show segments for healthy, provisioning, and degraded with those counts -- AND the chart center title SHALL show `8` +- THEN the donut chart SHALL show segments for healthy, provisioning, degraded, and failed with those counts +- AND the chart center title SHALL show `20` --- @@ -210,16 +233,16 @@ Version 1 of the operational dashboard SHALL distinguish **connected** metrics ( | Metric ID | Connected in v1 | Source when connected | | --- | --- | --- | -| `provisioned-gateways` | Yes | Gateway list aggregate (OP-DASH-06, OP-DASH-07) | -| `provisioned-sandboxes` | Yes | Sum of `active_sandbox_count` from gateway list | -| `registered-users` | Yes | See `platform/registered-users.spec.md` | +| `provisioned-gateways` | Yes | BFF `GET /api/metrics/gateways` (Prometheus); OP-DASH-23, OP-DASH-07 | +| `provisioned-sandboxes` | Yes | BFF `GET /api/metrics/gateway-sandboxes` (Prometheus); OP-DASH-06 | +| `registered-users` | Yes | BFF `GET /api/metrics/registered-users` (Prometheus); see `platform/registered-users.spec.md` | | `memory` | Yes | BFF `GET /api/metrics/cluster-memory` (Prometheus node-exporter); see `platform/cluster-memory.spec.md` | | `nodes` | Yes | BFF `GET /api/metrics/cluster-nodes` (Prometheus kube-state-metrics); see `platform/cluster-nodes.spec.md` | | `cpu` | Yes | BFF `GET /api/metrics/cluster-cpu` (Prometheus node-exporter); see `platform/cluster-cpu.spec.md` | | `pods` | Yes | BFF `GET /api/metrics/cluster-pods` (Prometheus kube-state-metrics); see `platform/cluster-pods.spec.md` | | `provision-time` | Yes | BFF `GET /api/metrics/gateway-provision-duration` (Prometheus control-plane histogram); see `platform/gateway-provision-time.spec.md` | -| `managed-clusters` | Yes | Paginated `GET /api/hypershell/v1/managed_clusters`; see `platform/platform-inventory.spec.md` | -| `managed-databases` | Yes | Paginated `GET /api/hypershell/v1/managed_databases`; see `platform/platform-inventory.spec.md` | +| `managed-clusters` | Yes | BFF `GET /api/metrics/platform-inventory` (Prometheus); see `platform/platform-inventory.spec.md` | +| `managed-databases` | Yes | BFF `GET /api/metrics/platform-inventory` (Prometheus); see `platform/platform-inventory.spec.md` | Widgets for placeholder metrics SHALL remain in the default layout and in the add-widgets drawer. When a metric ID is missing from the adapter response - whether because the metric is not yet connected or because its data source failed (OP-DASH-19) - the widget body SHALL render a localized "Metric unavailable" empty state (title and recovery guidance) instead of failing the entire dashboard. @@ -286,13 +309,13 @@ The host `DashboardControlPlane` adapter SHALL load operational metrics from ind | Source | Metric IDs affected | | --- | --- | -| Paginated gateway list (`GET /api/hypershell/v1/gateways`) | `provisioned-gateways`, `provisioned-sandboxes`, `provision-time` | -| Users list (`GET /api/hypershell/v1/users`, `page=1`, `size=1`) | `registered-users` | +| BFF `GET /api/metrics/gateways`, `GET /api/metrics/gateway-sandboxes`, and `GET /api/metrics/gateway-provision-duration` (`gateway-metrics`) | `provisioned-gateways`, `provisioned-sandboxes`, `provision-time` | +| BFF `GET /api/metrics/registered-users` (`registered-users`) | `registered-users` | | BFF `GET /api/metrics/cluster-memory` | `memory` | | BFF `GET /api/metrics/cluster-cpu` | `cpu` | | BFF `GET /api/metrics/cluster-pods` | `pods` | | BFF `GET /api/metrics/cluster-nodes` | `nodes` | -| Paginated managed cluster and managed database lists (`GET /api/hypershell/v1/managed_clusters`, `GET /api/hypershell/v1/managed_databases`) | `managed-clusters`, `managed-databases` | +| BFF `GET /api/metrics/platform-inventory` (`platform-inventory`) | `managed-clusters`, `managed-databases` | The adapter SHALL fetch these sources concurrently. When a source fails (network error, non-success HTTP status, inconsistent pagination, or other adapter validation error for that source), the adapter SHALL: @@ -311,19 +334,19 @@ Workflow probes for `get-operational-metrics` SHALL record outcome `succeeded` w The dashboard page SHALL derive partial-failure warnings from the adapter result (omitted expected metrics and/or explicit failure metadata) rather than treating a partial response as a query error that blocks the grid. -#### Scenario: Prometheus down does not hide gateway metrics +#### Scenario: Prometheus down does not hide unrelated metric sources -- GIVEN the gateway list and users list requests succeed -- AND every BFF cluster-metrics request fails +- GIVEN `GET /api/metrics/registered-users` succeeds +- AND every other BFF metrics request fails (gateway metrics, platform inventory, and cluster metrics) - WHEN the operator opens `/dashboard` -- THEN gateway, sandbox, and registered-user widgets SHALL display loaded values -- AND cluster metric widgets SHALL render the localized metric-unavailable state +- THEN the registered-user widget SHALL display loaded values +- AND gateway, sandbox, inventory, provision-time, and cluster metric widgets SHALL render the localized metric-unavailable state - AND a warning `Alert` SHALL explain that some metrics could not be loaded -#### Scenario: Gateway list failure does not hide cluster metrics +#### Scenario: Gateway metrics failure does not hide cluster metrics - GIVEN every BFF cluster-metrics request succeeds -- AND the gateway list request fails +- AND the `gateway-metrics` source fails (for example, `GET /api/metrics/gateways` returns HTTP `502`) - WHEN the operator opens `/dashboard` - THEN cluster metric widgets SHALL display loaded values - AND gateway, sandbox, and provision-time widgets or summary rows SHALL render the localized metric-unavailable state @@ -332,7 +355,7 @@ The dashboard page SHALL derive partial-failure warnings from the adapter result #### Scenario: Platform inventory failure does not hide other metrics - GIVEN every other metric source succeeds -- AND the managed cluster or managed database list request fails +- AND `GET /api/metrics/platform-inventory` fails - WHEN the operator opens `/dashboard` - THEN gateway, sandbox, registered-user, and cluster metric widgets SHALL display loaded values - AND inventory widgets and summary rows SHALL render the localized metric-unavailable state @@ -340,7 +363,8 @@ The dashboard page SHALL derive partial-failure warnings from the adapter result #### Scenario: No qualifying provision-time samples omit only provision time -- GIVEN the gateway list succeeds but contains no qualifying `Running` gateway duration samples +- GIVEN `GET /api/metrics/gateways` and `GET /api/metrics/gateway-sandboxes` succeed +- AND `GET /api/metrics/gateway-provision-duration` returns no qualifying histogram observations - WHEN `getOperationalMetrics` runs - THEN `provisioned-gateways` and `provisioned-sandboxes` SHALL still be emitted - AND `provision-time` SHALL be omitted From 0b5d435d8bd9750cf1e6977fb43da843216a2d98 Mon Sep 17 00:00:00 2001 From: Kim Doberstein Date: Fri, 11 Sep 2026 09:49:11 -0500 Subject: [PATCH 2/2] Code for leaning toward prometheus for dashboard --- .../api-server/pkg/rbac/user_provisioning.go | 10 +- components/api-server/plugins/gateways/dao.go | 15 + .../api-server/plugins/gateways/metrics.go | 33 + .../api-server/plugins/gateways/mock_dao.go | 8 + .../api-server/plugins/managedClusters/dao.go | 10 + .../plugins/managedClusters/inventory.go | 85 ++ .../plugins/managedClusters/inventory_test.go | 49 + .../plugins/managedClusters/metrics.go | 91 ++ .../plugins/managedClusters/mock_dao.go | 5 + .../plugins/managedClusters/plugin.go | 5 +- .../plugins/managedDatabases/dao.go | 9 + .../plugins/managedDatabases/inventory.go | 45 + .../managedDatabases/inventory_test.go | 21 + .../plugins/managedDatabases/metrics.go | 73 ++ .../plugins/managedDatabases/mock_dao.go | 4 + .../plugins/managedDatabases/plugin.go | 5 +- components/api-server/plugins/users/dao.go | 10 + .../api-server/plugins/users/metrics.go | 54 ++ components/api-server/plugins/users/plugin.go | 7 +- .../api/dashboard-control-plane.test.ts | 860 ++++++++---------- .../adapters/api/dashboard-control-plane.ts | 172 ++-- .../api/platform-inventory-aggregation.ts | 34 + .../require-dashboard-admin.test.tsx | 7 +- .../web-console/app/lib/session-roles.test.ts | 4 +- components/web-console/bff/src/app.ts | 66 ++ .../bff/src/metrics-gateway-sandboxes.ts | 26 + .../bff/src/metrics-platform-inventory.ts | 134 +++ .../bff/src/metrics-registered-users.ts | 25 + .../bff/src/prometheus-instant-query.ts | 129 +++ components/web-console/bff/src/roles.test.ts | 4 +- components/web-console/bff/test/auth.test.ts | 15 +- .../test/metrics-cluster-cpu-route.test.ts | 2 +- .../test/metrics-cluster-memory-route.test.ts | 2 +- .../test/metrics-cluster-nodes-route.test.ts | 2 +- .../test/metrics-cluster-pods-route.test.ts | 2 +- .../test/metrics-platform-inventory.test.ts | 224 +++++ .../web-console/shared/dashboard-roles.ts | 5 +- deploy/base/keycloak/keycloak.yaml | 2 +- .../src/gateways/gateway-data.test.ts | 18 + .../src/gateways/gateway-data.ts | 21 + packages/gateway-management-ui/src/index.ts | 1 + .../src/metrics/gateway-metrics-data.test.ts | 4 +- .../src/metrics/gateway-metrics-data.ts | 2 +- .../operational-dashboard-ui/DATA_SOURCES.md | 34 +- .../src/application/dashboard-types.ts | 2 +- .../src/dashboard/dashboard-metric-sources.ts | 4 +- 46 files changed, 1689 insertions(+), 651 deletions(-) create mode 100644 components/api-server/plugins/managedClusters/inventory.go create mode 100644 components/api-server/plugins/managedClusters/inventory_test.go create mode 100644 components/api-server/plugins/managedClusters/metrics.go create mode 100644 components/api-server/plugins/managedDatabases/inventory.go create mode 100644 components/api-server/plugins/managedDatabases/inventory_test.go create mode 100644 components/api-server/plugins/managedDatabases/metrics.go create mode 100644 components/api-server/plugins/users/metrics.go create mode 100644 components/web-console/bff/src/metrics-gateway-sandboxes.ts create mode 100644 components/web-console/bff/src/metrics-platform-inventory.ts create mode 100644 components/web-console/bff/src/metrics-registered-users.ts create mode 100644 components/web-console/bff/src/prometheus-instant-query.ts create mode 100644 components/web-console/bff/test/metrics-platform-inventory.test.ts diff --git a/components/api-server/pkg/rbac/user_provisioning.go b/components/api-server/pkg/rbac/user_provisioning.go index f2416b22c..acb22e29e 100644 --- a/components/api-server/pkg/rbac/user_provisioning.go +++ b/components/api-server/pkg/rbac/user_provisioning.go @@ -99,15 +99,7 @@ func HasPlatformAdminRole(ctx context.Context, userID string) bool { if userID == "" { return false } - v := ctx.Value(ContextJWTRolesKey) - if v == nil { - return false - } - jwtRoles, ok := v.([]string) - if !ok { - return false - } - for _, role := range jwtRoles { + for _, role := range GetJWTRolesFromContext(ctx) { if role == "platform:admin" { return true } diff --git a/components/api-server/plugins/gateways/dao.go b/components/api-server/plugins/gateways/dao.go index 779350984..bc116169f 100644 --- a/components/api-server/plugins/gateways/dao.go +++ b/components/api-server/plugins/gateways/dao.go @@ -35,6 +35,10 @@ type GatewayDao interface { SetActiveSandboxCount(ctx context.Context, namespace string, count int) (resulting int, err error) CountByPhase(ctx context.Context) (map[string]int64, error) + + // SumActiveSandboxCount returns the fleet-wide sum of active_sandbox_count + // across live gateways, treating NULL as zero. + SumActiveSandboxCount(ctx context.Context) (int64, error) } // sandboxCountRow captures the gateway identity and count returned by the @@ -230,3 +234,14 @@ func (d *sqlGatewayDao) CountByPhase(ctx context.Context) (map[string]int64, err } return counts, nil } + +func (d *sqlGatewayDao) SumActiveSandboxCount(ctx context.Context) (int64, error) { + g2 := (*d.sessionFactory).New(ctx) + var total int64 + if err := g2.Model(&Gateway{}). + Select("COALESCE(SUM(COALESCE(active_sandbox_count, 0)), 0)"). + Scan(&total).Error; err != nil { + return 0, err + } + return total, nil +} diff --git a/components/api-server/plugins/gateways/metrics.go b/components/api-server/plugins/gateways/metrics.go index e186707e9..01eb133b7 100644 --- a/components/api-server/plugins/gateways/metrics.go +++ b/components/api-server/plugins/gateways/metrics.go @@ -51,6 +51,7 @@ func RegisterGatewayMetrics(dao GatewayDao) { } prometheus.MustRegister(newGatewayCollector(dao)) + prometheus.MustRegister(newGatewayActiveSandboxesCollector(dao)) }) } @@ -101,3 +102,35 @@ func (c *gatewayCollector) Collect(ch chan<- prometheus.Metric) { } ch <- prometheus.MustNewConstMetric(c.desc, prometheus.GaugeValue, other, gatewayPhaseOther) } + +const activeSandboxesHelp = "Total active agent sandboxes across all gateways." + +type gatewayActiveSandboxesCollector struct { + dao GatewayDao + desc *prometheus.Desc +} + +func newGatewayActiveSandboxesCollector(dao GatewayDao) *gatewayActiveSandboxesCollector { + return &gatewayActiveSandboxesCollector{ + dao: dao, + desc: prometheus.NewDesc( + prometheus.BuildFQName(metricsNamespace, metricsSubsystem, "active_sandboxes_total"), + activeSandboxesHelp, + nil, + nil, + ), + } +} + +func (c *gatewayActiveSandboxesCollector) Describe(ch chan<- *prometheus.Desc) { + ch <- c.desc +} + +func (c *gatewayActiveSandboxesCollector) Collect(ch chan<- prometheus.Metric) { + total, err := c.dao.SumActiveSandboxCount(context.Background()) + if err != nil { + ch <- prometheus.NewInvalidMetric(c.desc, err) + return + } + ch <- prometheus.MustNewConstMetric(c.desc, prometheus.GaugeValue, float64(total)) +} diff --git a/components/api-server/plugins/gateways/mock_dao.go b/components/api-server/plugins/gateways/mock_dao.go index 186953395..c422e963b 100644 --- a/components/api-server/plugins/gateways/mock_dao.go +++ b/components/api-server/plugins/gateways/mock_dao.go @@ -87,6 +87,14 @@ func (d *gatewayDaoMock) CountByPhase(ctx context.Context) (map[string]int64, er return counts, nil } +func (d *gatewayDaoMock) SumActiveSandboxCount(ctx context.Context) (int64, error) { + var total int64 + for _, gw := range d.gateways { + total += int64(derefCount(gw.ActiveSandboxCount)) + } + return total, nil +} + func (d *gatewayDaoMock) findByNamespace(namespace string) *Gateway { for _, gateway := range d.gateways { if gateway.Namespace == namespace { diff --git a/components/api-server/plugins/managedClusters/dao.go b/components/api-server/plugins/managedClusters/dao.go index f27775552..eda5aa494 100644 --- a/components/api-server/plugins/managedClusters/dao.go +++ b/components/api-server/plugins/managedClusters/dao.go @@ -2,6 +2,7 @@ package managedClusters import ( "context" + "time" "gorm.io/gorm/clause" @@ -17,6 +18,7 @@ type ManagedClusterDao interface { FindByIDs(ctx context.Context, ids []string) (ManagedClusterList, error) All(ctx context.Context) (ManagedClusterList, error) FindByOIDCSubject(ctx context.Context, subject string) (*ManagedCluster, error) + InventorySnapshot(ctx context.Context, evaluationTime time.Time) (*ClusterInventorySnapshot, error) } var _ ManagedClusterDao = &sqlManagedClusterDao{} @@ -91,3 +93,11 @@ func (d *sqlManagedClusterDao) FindByOIDCSubject(ctx context.Context, subject st } return &managedCluster, nil } + +func (d *sqlManagedClusterDao) InventorySnapshot(ctx context.Context, evaluationTime time.Time) (*ClusterInventorySnapshot, error) { + clusters, err := d.All(ctx) + if err != nil { + return nil, err + } + return buildClusterInventorySnapshot(clusters, evaluationTime), nil +} diff --git a/components/api-server/plugins/managedClusters/inventory.go b/components/api-server/plugins/managedClusters/inventory.go new file mode 100644 index 000000000..91fdbe6b3 --- /dev/null +++ b/components/api-server/plugins/managedClusters/inventory.go @@ -0,0 +1,85 @@ +package managedClusters + +import ( + "strings" + "time" +) + +const inventoryLookback = 30 * 24 * time.Hour + +// ClusterInventoryRow is one labeled bucket in the managed cluster inventory +// aggregate exposed to Prometheus. +type ClusterInventoryRow struct { + Status string + Provider string + Region string + Count int64 +} + +// ClusterInventorySnapshot is the fleet-wide managed cluster inventory computed +// on each metrics scrape. +type ClusterInventorySnapshot struct { + CreatedLast30Days int64 + Rows []ClusterInventoryRow +} + +func inventoryBucketString(value string) string { + trimmed := strings.TrimSpace(value) + if trimmed == "" { + return "unknown" + } + return trimmed +} + +func inventoryBucketOptional(value *string) string { + if value == nil { + return "unknown" + } + return inventoryBucketString(*value) +} + +func inventoryLookbackStart(evaluationTime time.Time) time.Time { + return evaluationTime.UTC().Add(-inventoryLookback) +} + +func buildClusterInventorySnapshot( + clusters ManagedClusterList, + evaluationTime time.Time, +) *ClusterInventorySnapshot { + windowStart := inventoryLookbackStart(evaluationTime) + type inventoryKey struct { + status string + provider string + region string + } + counts := make(map[inventoryKey]int64) + var createdLast30Days int64 + + for _, cluster := range clusters { + key := inventoryKey{ + status: inventoryBucketOptional(cluster.Status), + provider: inventoryBucketString(cluster.Provider), + region: inventoryBucketOptional(cluster.Region), + } + counts[key]++ + + if !cluster.CreatedAt.IsZero() && !cluster.CreatedAt.Before(windowStart) { + createdLast30Days++ + } + } + + rows := make([]ClusterInventoryRow, 0, len(counts)) + for key, count := range counts { + rows = append(rows, ClusterInventoryRow{ + Status: key.status, + Provider: key.provider, + Region: key.region, + Count: count, + }) + } + + return &ClusterInventorySnapshot{ + CreatedLast30Days: createdLast30Days, + Rows: rows, + } +} diff --git a/components/api-server/plugins/managedClusters/inventory_test.go b/components/api-server/plugins/managedClusters/inventory_test.go new file mode 100644 index 000000000..4fb16d743 --- /dev/null +++ b/components/api-server/plugins/managedClusters/inventory_test.go @@ -0,0 +1,49 @@ +package managedClusters + +import ( + "testing" + "time" + + "github.com/openshift-online/rh-trex-ai/pkg/api" +) + +func TestBuildClusterInventorySnapshot(t *testing.T) { + evaluationTime := time.Date(2026, 9, 10, 12, 0, 0, 0, time.UTC) + recentCreatedAt := evaluationTime.Add(-10 * 24 * time.Hour) + oldCreatedAt := evaluationTime.Add(-60 * 24 * time.Hour) + + clusters := ManagedClusterList{ + { + Meta: api.Meta{CreatedAt: recentCreatedAt}, + Provider: "aws", + Region: strPtr("us-east-1"), + Status: strPtr("Ready"), + }, + { + Meta: api.Meta{CreatedAt: oldCreatedAt}, + Provider: "openshift", + Region: nil, + Status: strPtr("Failed"), + }, + { + Meta: api.Meta{CreatedAt: recentCreatedAt}, + Provider: "aws", + Region: strPtr(""), + Status: nil, + }, + } + + snapshot := buildClusterInventorySnapshot(clusters, evaluationTime) + + if snapshot.CreatedLast30Days != 2 { + t.Fatalf("expected 2 clusters created in the last 30 days, got %d", snapshot.CreatedLast30Days) + } + + if len(snapshot.Rows) != 3 { + t.Fatalf("expected 3 inventory rows, got %d", len(snapshot.Rows)) + } +} + +func strPtr(value string) *string { + return &value +} diff --git a/components/api-server/plugins/managedClusters/metrics.go b/components/api-server/plugins/managedClusters/metrics.go new file mode 100644 index 000000000..91fb49d55 --- /dev/null +++ b/components/api-server/plugins/managedClusters/metrics.go @@ -0,0 +1,91 @@ +package managedClusters + +import ( + "context" + "sync" + "time" + + "github.com/prometheus/client_golang/prometheus" +) + +const ( + metricsNamespace = "hypershell" + metricsSubsystem = "managed_clusters" +) + +var ( + managedClusterMetricsOnce sync.Once +) + +// RegisterManagedClusterMetrics registers Prometheus gauges for fleet-wide +// managed cluster inventory. Safe to call multiple times. +func RegisterManagedClusterMetrics(dao ManagedClusterDao) { + managedClusterMetricsOnce.Do(func() { + prometheus.MustRegister(newManagedClusterInventoryCollector(dao)) + }) +} + +type managedClusterInventoryCollector struct { + dao ManagedClusterDao + totalDesc *prometheus.Desc + createdDesc *prometheus.Desc + dimensionedDesc *prometheus.Desc +} + +func newManagedClusterInventoryCollector(dao ManagedClusterDao) *managedClusterInventoryCollector { + return &managedClusterInventoryCollector{ + dao: dao, + totalDesc: prometheus.NewDesc( + prometheus.BuildFQName(metricsNamespace, metricsSubsystem, "total"), + "Total registered managed clusters.", + nil, + nil, + ), + createdDesc: prometheus.NewDesc( + prometheus.BuildFQName(metricsNamespace, metricsSubsystem, "created_last_30_days_total"), + "Managed clusters created in the last 30 days.", + nil, + nil, + ), + dimensionedDesc: prometheus.NewDesc( + prometheus.BuildFQName(metricsNamespace, metricsSubsystem, "inventory_total"), + "Managed clusters by inventory status, provider, and region.", + []string{"status", "provider", "region"}, + nil, + ), + } +} + +func (c *managedClusterInventoryCollector) Describe(ch chan<- *prometheus.Desc) { + ch <- c.totalDesc + ch <- c.createdDesc + ch <- c.dimensionedDesc +} + +func (c *managedClusterInventoryCollector) Collect(ch chan<- prometheus.Metric) { + snapshot, err := c.dao.InventorySnapshot(context.Background(), time.Now().UTC()) + if err != nil { + ch <- prometheus.NewInvalidMetric(c.totalDesc, err) + return + } + + var total int64 + for _, row := range snapshot.Rows { + total += row.Count + ch <- prometheus.MustNewConstMetric( + c.dimensionedDesc, + prometheus.GaugeValue, + float64(row.Count), + row.Status, + row.Provider, + row.Region, + ) + } + + ch <- prometheus.MustNewConstMetric(c.totalDesc, prometheus.GaugeValue, float64(total)) + ch <- prometheus.MustNewConstMetric( + c.createdDesc, + prometheus.GaugeValue, + float64(snapshot.CreatedLast30Days), + ) +} diff --git a/components/api-server/plugins/managedClusters/mock_dao.go b/components/api-server/plugins/managedClusters/mock_dao.go index 1111c1a4d..c4bea9333 100644 --- a/components/api-server/plugins/managedClusters/mock_dao.go +++ b/components/api-server/plugins/managedClusters/mock_dao.go @@ -2,6 +2,7 @@ package managedClusters import ( "context" + "time" "gorm.io/gorm" @@ -62,3 +63,7 @@ func (d *managedClusterDaoMock) FindByOIDCSubject(ctx context.Context, subject s } return nil, gorm.ErrRecordNotFound } + +func (d *managedClusterDaoMock) InventorySnapshot(ctx context.Context, evaluationTime time.Time) (*ClusterInventorySnapshot, error) { + return buildClusterInventorySnapshot(d.managedClusters, evaluationTime), nil +} diff --git a/components/api-server/plugins/managedClusters/plugin.go b/components/api-server/plugins/managedClusters/plugin.go index ff7aeb000..5f1ed5e78 100644 --- a/components/api-server/plugins/managedClusters/plugin.go +++ b/components/api-server/plugins/managedClusters/plugin.go @@ -22,10 +22,13 @@ import ( type ServiceLocator func() ManagedClusterService func NewServiceLocator(env *environments.Env) ServiceLocator { + dao := NewManagedClusterDao(&env.Database.SessionFactory) + RegisterManagedClusterMetrics(dao) + return func() ManagedClusterService { return NewManagedClusterService( db.NewAdvisoryLockFactory(env.Database.SessionFactory), - NewManagedClusterDao(&env.Database.SessionFactory), + dao, events.Service(&env.Services), ) } diff --git a/components/api-server/plugins/managedDatabases/dao.go b/components/api-server/plugins/managedDatabases/dao.go index 2326fecbf..ace807d7c 100644 --- a/components/api-server/plugins/managedDatabases/dao.go +++ b/components/api-server/plugins/managedDatabases/dao.go @@ -19,6 +19,7 @@ type ManagedDatabaseDao interface { FindByIDs(ctx context.Context, ids []string) (ManagedDatabaseList, error) All(ctx context.Context) (ManagedDatabaseList, error) ExistsByDatabaseID(ctx context.Context, databaseID string) (bool, error) + InventorySnapshot(ctx context.Context) (*DatabaseInventorySnapshot, error) } var _ ManagedDatabaseDao = &sqlManagedDatabaseDao{} @@ -116,3 +117,11 @@ func (d *sqlManagedDatabaseDao) ExistsByDatabaseID(ctx context.Context, database } return count > 0, nil } + +func (d *sqlManagedDatabaseDao) InventorySnapshot(ctx context.Context) (*DatabaseInventorySnapshot, error) { + databases, err := d.All(ctx) + if err != nil { + return nil, err + } + return buildDatabaseInventorySnapshot(databases), nil +} diff --git a/components/api-server/plugins/managedDatabases/inventory.go b/components/api-server/plugins/managedDatabases/inventory.go new file mode 100644 index 000000000..6ebaa9d7b --- /dev/null +++ b/components/api-server/plugins/managedDatabases/inventory.go @@ -0,0 +1,45 @@ +package managedDatabases + +import "strings" + +// DatabaseInventoryRow is one status bucket in the managed database inventory +// aggregate exposed to Prometheus. +type DatabaseInventoryRow struct { + Status string + Count int64 +} + +// DatabaseInventorySnapshot is the fleet-wide managed database inventory +// computed on each metrics scrape. +type DatabaseInventorySnapshot struct { + Rows []DatabaseInventoryRow +} + +func inventoryBucketOptional(value *string) string { + if value == nil { + return "unknown" + } + trimmed := strings.TrimSpace(*value) + if trimmed == "" { + return "unknown" + } + return trimmed +} + +func buildDatabaseInventorySnapshot(databases ManagedDatabaseList) *DatabaseInventorySnapshot { + counts := make(map[string]int64) + for _, database := range databases { + status := inventoryBucketOptional(database.Status) + counts[status]++ + } + + rows := make([]DatabaseInventoryRow, 0, len(counts)) + for status, count := range counts { + rows = append(rows, DatabaseInventoryRow{ + Status: status, + Count: count, + }) + } + + return &DatabaseInventorySnapshot{Rows: rows} +} diff --git a/components/api-server/plugins/managedDatabases/inventory_test.go b/components/api-server/plugins/managedDatabases/inventory_test.go new file mode 100644 index 000000000..d3d394e8b --- /dev/null +++ b/components/api-server/plugins/managedDatabases/inventory_test.go @@ -0,0 +1,21 @@ +package managedDatabases + +import "testing" + +func TestBuildDatabaseInventorySnapshot(t *testing.T) { + databases := ManagedDatabaseList{ + {Status: strPtr("Ready")}, + {Status: nil}, + {Status: strPtr("")}, + } + + snapshot := buildDatabaseInventorySnapshot(databases) + + if len(snapshot.Rows) != 2 { + t.Fatalf("expected 2 inventory rows, got %d", len(snapshot.Rows)) + } +} + +func strPtr(value string) *string { + return &value +} diff --git a/components/api-server/plugins/managedDatabases/metrics.go b/components/api-server/plugins/managedDatabases/metrics.go new file mode 100644 index 000000000..fdedb2ee6 --- /dev/null +++ b/components/api-server/plugins/managedDatabases/metrics.go @@ -0,0 +1,73 @@ +package managedDatabases + +import ( + "context" + "sync" + + "github.com/prometheus/client_golang/prometheus" +) + +const ( + metricsNamespace = "hypershell" + metricsSubsystem = "managed_databases" +) + +var managedDatabaseMetricsOnce sync.Once + +// RegisterManagedDatabaseMetrics registers Prometheus gauges for fleet-wide +// managed database inventory. Safe to call multiple times. +func RegisterManagedDatabaseMetrics(dao ManagedDatabaseDao) { + managedDatabaseMetricsOnce.Do(func() { + prometheus.MustRegister(newManagedDatabaseInventoryCollector(dao)) + }) +} + +type managedDatabaseInventoryCollector struct { + dao ManagedDatabaseDao + totalDesc *prometheus.Desc + dimensionedDesc *prometheus.Desc +} + +func newManagedDatabaseInventoryCollector(dao ManagedDatabaseDao) *managedDatabaseInventoryCollector { + return &managedDatabaseInventoryCollector{ + dao: dao, + totalDesc: prometheus.NewDesc( + prometheus.BuildFQName(metricsNamespace, metricsSubsystem, "total"), + "Total registered managed databases.", + nil, + nil, + ), + dimensionedDesc: prometheus.NewDesc( + prometheus.BuildFQName(metricsNamespace, metricsSubsystem, "inventory_total"), + "Managed databases by inventory status.", + []string{"status"}, + nil, + ), + } +} + +func (c *managedDatabaseInventoryCollector) Describe(ch chan<- *prometheus.Desc) { + ch <- c.totalDesc + ch <- c.dimensionedDesc +} + +func (c *managedDatabaseInventoryCollector) Collect(ch chan<- prometheus.Metric) { + snapshot, err := c.dao.InventorySnapshot(context.Background()) + if err != nil { + ch <- prometheus.NewInvalidMetric(c.totalDesc, err) + return + } + + var total int64 + for _, row := range snapshot.Rows { + total += row.Count + ch <- prometheus.MustNewConstMetric( + c.dimensionedDesc, + prometheus.GaugeValue, + float64(row.Count), + row.Status, + ) + } + + ch <- prometheus.MustNewConstMetric(c.totalDesc, prometheus.GaugeValue, float64(total)) +} diff --git a/components/api-server/plugins/managedDatabases/mock_dao.go b/components/api-server/plugins/managedDatabases/mock_dao.go index 37b3d9269..e98517021 100644 --- a/components/api-server/plugins/managedDatabases/mock_dao.go +++ b/components/api-server/plugins/managedDatabases/mock_dao.go @@ -59,3 +59,7 @@ func (d *managedDatabaseDaoMock) All(ctx context.Context) (ManagedDatabaseList, func (d *managedDatabaseDaoMock) ExistsByDatabaseID(ctx context.Context, databaseID string) (bool, error) { return false, nil } + +func (d *managedDatabaseDaoMock) InventorySnapshot(ctx context.Context) (*DatabaseInventorySnapshot, error) { + return buildDatabaseInventorySnapshot(d.managedDatabases), nil +} diff --git a/components/api-server/plugins/managedDatabases/plugin.go b/components/api-server/plugins/managedDatabases/plugin.go index 28a1c63c2..d8aea6081 100644 --- a/components/api-server/plugins/managedDatabases/plugin.go +++ b/components/api-server/plugins/managedDatabases/plugin.go @@ -22,10 +22,13 @@ import ( type ServiceLocator func() ManagedDatabaseService func NewServiceLocator(env *environments.Env) ServiceLocator { + dao := NewManagedDatabaseDao(&env.Database.SessionFactory) + RegisterManagedDatabaseMetrics(dao) + return func() ManagedDatabaseService { return NewManagedDatabaseService( db.NewAdvisoryLockFactory(env.Database.SessionFactory), - NewManagedDatabaseDao(&env.Database.SessionFactory), + dao, events.Service(&env.Services), ) } diff --git a/components/api-server/plugins/users/dao.go b/components/api-server/plugins/users/dao.go index 867257ff7..90a18b891 100644 --- a/components/api-server/plugins/users/dao.go +++ b/components/api-server/plugins/users/dao.go @@ -18,6 +18,7 @@ type UserDao interface { Upsert(ctx context.Context, user *User) (*User, error) FindByIDs(ctx context.Context, ids []string) (UserList, error) All(ctx context.Context) (UserList, error) + CountRegistered(ctx context.Context) (int64, error) } var _ UserDao = &sqlUserDao{} @@ -112,3 +113,12 @@ func (d *sqlUserDao) All(ctx context.Context) (UserList, error) { } return users, nil } + +func (d *sqlUserDao) CountRegistered(ctx context.Context) (int64, error) { + g2 := (*d.sessionFactory).New(ctx) + var count int64 + if err := g2.Model(&User{}).Count(&count).Error; err != nil { + return 0, err + } + return count, nil +} diff --git a/components/api-server/plugins/users/metrics.go b/components/api-server/plugins/users/metrics.go new file mode 100644 index 000000000..88ad4c9dd --- /dev/null +++ b/components/api-server/plugins/users/metrics.go @@ -0,0 +1,54 @@ +package users + +import ( + "context" + "sync" + + "github.com/prometheus/client_golang/prometheus" +) + +const ( + metricsNamespace = "hypershell" + metricsSubsystem = "users" +) + +var userMetricsOnce sync.Once + +// RegisterUserMetrics registers Prometheus gauges for registered user counts. +// Safe to call multiple times. +func RegisterUserMetrics(dao UserDao) { + userMetricsOnce.Do(func() { + prometheus.MustRegister(newRegisteredUsersCollector(dao)) + }) +} + +type registeredUsersCollector struct { + dao UserDao + desc *prometheus.Desc +} + +func newRegisteredUsersCollector(dao UserDao) *registeredUsersCollector { + return ®isteredUsersCollector{ + dao: dao, + desc: prometheus.NewDesc( + prometheus.BuildFQName(metricsNamespace, metricsSubsystem, "registered_total"), + "Total registered users.", + nil, + nil, + ), + } +} + +func (c *registeredUsersCollector) Describe(ch chan<- *prometheus.Desc) { + ch <- c.desc +} + +func (c *registeredUsersCollector) Collect(ch chan<- prometheus.Metric) { + count, err := c.dao.CountRegistered(context.Background()) + if err != nil { + ch <- prometheus.NewInvalidMetric(c.desc, err) + return + } + + ch <- prometheus.MustNewConstMetric(c.desc, prometheus.GaugeValue, float64(count)) +} diff --git a/components/api-server/plugins/users/plugin.go b/components/api-server/plugins/users/plugin.go index 3e4b00012..6ab38a1ee 100644 --- a/components/api-server/plugins/users/plugin.go +++ b/components/api-server/plugins/users/plugin.go @@ -17,10 +17,11 @@ import ( type ServiceLocator func() UserService func NewServiceLocator(env *environments.Env) ServiceLocator { + dao := NewUserDao(&env.Database.SessionFactory) + RegisterUserMetrics(dao) + return func() UserService { - return NewUserService( - NewUserDao(&env.Database.SessionFactory), - ) + return NewUserService(dao) } } diff --git a/components/web-console/app/adapters/api/dashboard-control-plane.test.ts b/components/web-console/app/adapters/api/dashboard-control-plane.test.ts index acf7e31cc..e648b24db 100644 --- a/components/web-console/app/adapters/api/dashboard-control-plane.test.ts +++ b/components/web-console/app/adapters/api/dashboard-control-plane.test.ts @@ -1,37 +1,11 @@ -import type { Gateway, GatewayList } from "@openshift-online/hypershell-sdk"; -import type { - ManagedCluster, - ManagedClusterList, - ManagedDatabase, - ManagedDatabaseList, -} from "@openshift-online/hypershell-sdk"; import type { SDKClient } from "@openshift-online/hypershell-sdk"; import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; import { createDashboardControlPlaneAdapter } from "./dashboard-control-plane"; +import type { PlatformInventoryMetricsResponse } from "./platform-inventory-aggregation"; -const gatewayListApi = vi.fn(); -const usersListApi = vi.fn(); -const managedClustersListApi = vi.fn(); -const managedDatabasesListApi = vi.fn(); const fetchMock = vi.fn(); -const apiFactory = vi.fn( - () => - ({ - gateways: { - list: gatewayListApi, - }, - managedClusters: { - list: managedClustersListApi, - }, - managedDatabases: { - list: managedDatabasesListApi, - }, - users: { - list: usersListApi, - }, - }) as unknown as SDKClient, -); +const apiFactory = vi.fn(() => ({}) as unknown as SDKClient); const adapter = createDashboardControlPlaneAdapter(apiFactory); const context = { @@ -56,11 +30,71 @@ const mockGatewayProvisionDurationResponse = { p95_seconds: 726, }; +const defaultGatewayPhaseCounts = { + Running: 100, + Provisioning: 50, +}; + +function defaultPlatformInventory(): PlatformInventoryMetricsResponse { + return { + managed_clusters: { + by_provider: {}, + by_region: {}, + by_status: {}, + created_last_30_days: 0, + total: 0, + }, + managed_databases: { + by_status: {}, + total: 0, + }, + }; +} + +interface DashboardMetricsMockOptions { + activeSandboxes?: number; + omitProvisionDuration?: boolean; + platformInventory?: PlatformInventoryMetricsResponse; + registeredUsers?: { total_registered: number }; +} + function mockClusterMetricsResponses( capacityBytes: number, usedBytes: number, + gatewayPhaseCounts: Record = defaultGatewayPhaseCounts, + options: DashboardMetricsMockOptions = {}, ): void { + const activeSandboxes = options.activeSandboxes ?? 0; + const registeredUsers = + options.registeredUsers ?? defaultRegisteredUsersResponse(0); + const platformInventory = + options.platformInventory ?? defaultPlatformInventory(); + fetchMock.mockImplementation((url: string) => { + if (url === "/api/metrics/gateways") { + return Promise.resolve({ + json: () => Promise.resolve({ counts: gatewayPhaseCounts }), + ok: true, + }); + } + if (url === "/api/metrics/gateway-sandboxes") { + return Promise.resolve({ + json: () => Promise.resolve({ active_sandboxes: activeSandboxes }), + ok: true, + }); + } + if (url === "/api/metrics/registered-users") { + return Promise.resolve({ + json: () => Promise.resolve(registeredUsers), + ok: true, + }); + } + if (url === "/api/metrics/platform-inventory") { + return Promise.resolve({ + json: () => Promise.resolve(platformInventory), + ok: true, + }); + } if (url === "/api/metrics/cluster-memory") { return Promise.resolve({ json: () => @@ -101,6 +135,12 @@ function mockClusterMetricsResponses( }); } if (url === "/api/metrics/gateway-provision-duration") { + if (options.omitProvisionDuration) { + return Promise.resolve({ + ok: false, + status: 502, + }); + } return Promise.resolve({ json: () => Promise.resolve(mockGatewayProvisionDurationResponse), ok: true, @@ -110,207 +150,76 @@ function mockClusterMetricsResponses( }); } +function resolveStandardPrometheusSupportRoutes( + url: string, + options: DashboardMetricsMockOptions = {}, +) { + const activeSandboxes = options.activeSandboxes ?? 0; + const registeredUsers = + options.registeredUsers ?? defaultRegisteredUsersResponse(0); + const platformInventory = + options.platformInventory ?? defaultPlatformInventory(); + + if (url === "/api/metrics/gateway-sandboxes") { + return Promise.resolve({ + json: () => Promise.resolve({ active_sandboxes: activeSandboxes }), + ok: true, + }); + } + if (url === "/api/metrics/registered-users") { + return Promise.resolve({ + json: () => Promise.resolve(registeredUsers), + ok: true, + }); + } + if (url === "/api/metrics/platform-inventory") { + return Promise.resolve({ + json: () => Promise.resolve(platformInventory), + ok: true, + }); + } + if (url === "/api/metrics/gateway-provision-duration") { + if (options.omitProvisionDuration) { + return Promise.resolve({ + ok: false, + status: 502, + }); + } + return Promise.resolve({ + json: () => Promise.resolve(mockGatewayProvisionDurationResponse), + ok: true, + }); + } + return undefined; +} + beforeEach(() => { vi.stubGlobal("fetch", fetchMock); fetchMock.mockReset(); - gatewayListApi.mockReset(); - usersListApi.mockReset(); - managedClustersListApi.mockReset(); - managedDatabasesListApi.mockReset(); - managedClustersListApi.mockResolvedValue({ - items: [], - kind: "ManagedClusterList", - page: 1, - size: 0, - total: 0, - }); - managedDatabasesListApi.mockResolvedValue({ - items: [], - kind: "ManagedDatabaseList", - page: 1, - size: 0, - total: 0, - }); }); afterEach(() => { vi.unstubAllGlobals(); }); -function gateway(overrides: Partial = {}): Gateway { - const phase = overrides.phase ?? "Running"; - const runningTimestamps = - phase === "Running" - ? { - created_at: "2026-08-01T10:00:00.000Z", - updated_at: "2026-08-01T10:05:15.000Z", - } - : { - created_at: null, - updated_at: null, - }; - - return { - active_sandbox_count: 2, - cluster_id: "", - console_address: "", - created_by: "", - credential_driver: "", - database_id: "database-1", - external_dns: "gateway.example.com", - href: "/api/hypershell/v1/gateways/gateway-1", - id: "gateway-1", - image: "", - kind: "Gateway", - name: "Team gateway", - namespace: "openshell", - oidc: "", - phase: "Running", - release_id: "release-1", - route: "", - route_address: "", - server_dns_names: "", - service_type: "", - status: "Healthy", - supervisor_image: "", - tls_mode: "", - ...runningTimestamps, - ...overrides, - }; -} - -function gatewayList( - items: Gateway[], - total = items.length, - page = 1, -): GatewayList { - return { - items, - kind: "GatewayList", - page, - size: items.length, - total, - }; -} - -function managedCluster( - overrides: Partial = {}, -): ManagedCluster { - return { - api_server_url: "", - created_at: "2026-08-01T10:00:00.000Z", - href: "/api/hypershell/v1/managed_clusters/cluster-1", - id: "cluster-1", - kind: "ManagedCluster", - kubeconfig_secret: "secret", - last_seen_at: "", - name: "cluster-1", - oidc_subject: "", - provider: "aws", - region: "us-east-1", - status: "Ready", - updated_at: "2026-08-01T10:00:00.000Z", - ...overrides, - }; -} - -function managedClusterList( - items: ManagedCluster[], - total = items.length, - page = 1, -): ManagedClusterList { - return { - items, - kind: "ManagedClusterList", - page, - size: items.length, - total, - }; -} - -function managedDatabase( - overrides: Partial = {}, -): ManagedDatabase { - return { - connection_secret: "secret", - created_at: "2026-08-01T10:00:00.000Z", - engine: "postgres", - engine_version: "16", - href: "/api/hypershell/v1/managed_databases/database-1", - id: "database-1", - instance_class: "small", - kind: "ManagedDatabase", - name: "database-1", - namespace: "openshell", - provider: "aws", - region: "us-east-1", - status: "Ready", - updated_at: "2026-08-01T10:00:00.000Z", - ...overrides, - }; -} - -function managedDatabaseList( - items: ManagedDatabase[], - total = items.length, - page = 1, -): ManagedDatabaseList { - return { - items, - kind: "ManagedDatabaseList", - page, - size: items.length, - total, - }; +function defaultRegisteredUsersResponse(total = 0) { + return { total_registered: total }; } describe("createDashboardControlPlaneAdapter", () => { - it("aggregates paginated gateway lists into operational metrics", async () => { - mockClusterMetricsResponses(254468212736, 236223201280); - - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 42, - }); - - const firstPage = Array.from({ length: 100 }, (_, index) => - gateway({ - active_sandbox_count: 1, - id: `gateway-${String(index)}`, - phase: "Running", - status: "Healthy", - }), - ); - const secondPage = Array.from({ length: 50 }, (_, index) => - gateway({ - active_sandbox_count: 2, - id: `gateway-${String(index + 100)}`, - phase: "Provisioning", - status: "route pending", - }), + it("aggregates Prometheus dashboard metrics into operational metrics", async () => { + mockClusterMetricsResponses( + 254468212736, + 236223201280, + { Provisioning: 50, Running: 100 }, + { + activeSandboxes: 200, + registeredUsers: { total_registered: 42 }, + }, ); - gatewayListApi - .mockResolvedValueOnce(gatewayList(firstPage, 150, 1)) - .mockResolvedValueOnce(gatewayList(secondPage, 150, 2)); - const metrics = await adapter.getOperationalMetrics(context); - expect(gatewayListApi).toHaveBeenCalledTimes(2); - expect(gatewayListApi).toHaveBeenNthCalledWith( - 1, - { orderBy: "name asc", page: 1, size: 100 }, - { signal: undefined }, - ); - expect(gatewayListApi).toHaveBeenNthCalledWith( - 2, - { orderBy: "name asc", page: 2, size: 100 }, - { signal: undefined }, - ); - const gatewaysMetric = metrics.metrics.find( (metric) => metric.id === "provisioned-gateways", ); @@ -338,7 +247,10 @@ describe("createDashboardControlPlaneAdapter", () => { provisioning: 50, }); expect(sandboxesMetric?.value).toBe("200"); - expect(registeredUsersMetric?.value).toBe("42"); + expect(registeredUsersMetric).toEqual({ + id: "registered-users", + value: "42", + }); expect(memoryMetric).toEqual({ id: "memory", total: "237", @@ -398,40 +310,33 @@ describe("createDashboardControlPlaneAdapter", () => { credentials: "same-origin", signal: undefined, }); - expect(usersListApi).toHaveBeenCalledWith( - { orderBy: "username asc", page: 1, size: 1 }, - { signal: undefined }, - ); + expect(fetchMock).toHaveBeenCalledWith("/api/metrics/gateways", { + credentials: "same-origin", + signal: undefined, + }); + expect(fetchMock).toHaveBeenCalledWith("/api/metrics/gateway-sandboxes", { + credentials: "same-origin", + signal: undefined, + }); + expect(fetchMock).toHaveBeenCalledWith("/api/metrics/registered-users", { + credentials: "same-origin", + signal: undefined, + }); }); - it("maps gateway lifecycle fields into display-status buckets", async () => { - mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2); - - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, + it("maps Prometheus gateway phase counts into display-status buckets", async () => { + mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2, { + Running: 1, + Degraded: 1, + Failed: 1, }); - gatewayListApi.mockResolvedValueOnce( - gatewayList( - [ - gateway({ phase: "Running", status: "Healthy" }), - gateway({ phase: "Degraded", status: "CrashLoopBackOff" }), - gateway({ phase: "Failed", status: "apply error" }), - ], - 3, - 1, - ), - ); - const metrics = await adapter.getOperationalMetrics(context); const gatewaysMetric = metrics.metrics.find( (metric) => metric.id === "provisioned-gateways", ); + expect(gatewaysMetric?.value).toBe("3"); expect(gatewaysMetric?.status).toEqual({ degraded: 1, failed: 1, @@ -440,27 +345,14 @@ describe("createDashboardControlPlaneAdapter", () => { }); }); - it("treats omitted active_sandbox_count as zero when summing sandboxes", async () => { - mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2); - - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - - gatewayListApi.mockResolvedValueOnce( - gatewayList( - [ - gateway({ active_sandbox_count: 2 }), - gateway({ active_sandbox_count: undefined }), - gateway({ active_sandbox_count: 3 }), - ], - 3, - 1, - ), + it("loads sandbox totals from Prometheus", async () => { + mockClusterMetricsResponses( + 1024 ** 3, + 512 * 1024 ** 2, + defaultGatewayPhaseCounts, + { + activeSandboxes: 5, + }, ); const metrics = await adapter.getOperationalMetrics(context); @@ -474,22 +366,6 @@ describe("createDashboardControlPlaneAdapter", () => { it("maps gateway provision duration histogram into average, P50, and P95 minutes", async () => { mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2); - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - - gatewayListApi.mockResolvedValueOnce( - gatewayList( - [gateway({ phase: "Provisioning", status: "route pending" })], - 1, - 1, - ), - ); - const metrics = await adapter.getOperationalMetrics(context); const provisionTimeMetric = metrics.metrics.find( (metric) => metric.id === "provision-time", @@ -508,11 +384,61 @@ describe("createDashboardControlPlaneAdapter", () => { }); it("omits provision time when the BFF provision duration route is unavailable", async () => { + mockClusterMetricsResponses( + 1024 ** 3, + 512 * 1024 ** 2, + defaultGatewayPhaseCounts, + { + omitProvisionDuration: true, + }, + ); + + const metrics = await adapter.getOperationalMetrics(context); + const provisionTimeMetric = metrics.metrics.find( + (metric) => metric.id === "provision-time", + ); + const memoryMetric = metrics.metrics.find( + (metric) => metric.id === "memory", + ); + + expect(provisionTimeMetric).toBeUndefined(); + expect(memoryMetric).toEqual({ + id: "memory", + total: "1", + unit: "GiB", + value: "1", + }); + expect( + metrics.metrics.find((metric) => metric.id === "provisioned-gateways"), + ).toBeDefined(); + }); + + it("omits gateway metrics when the gateway-sandboxes route fails", async () => { fetchMock.mockImplementation((url: string) => { - if (url === "/api/metrics/gateway-provision-duration") { + const base = { + activeSandboxes: 0, + registeredUsers: defaultRegisteredUsersResponse(0), + platformInventory: defaultPlatformInventory(), + }; + if (url === "/api/metrics/gateway-sandboxes") { + return Promise.resolve({ ok: false, status: 502 }); + } + if (url === "/api/metrics/gateways") { return Promise.resolve({ - ok: false, - status: 502, + json: () => Promise.resolve({ counts: defaultGatewayPhaseCounts }), + ok: true, + }); + } + if (url === "/api/metrics/registered-users") { + return Promise.resolve({ + json: () => Promise.resolve(base.registeredUsers), + ok: true, + }); + } + if (url === "/api/metrics/platform-inventory") { + return Promise.resolve({ + json: () => Promise.resolve(base.platformInventory), + ok: true, }); } if (url === "/api/metrics/cluster-memory") { @@ -554,63 +480,19 @@ describe("createDashboardControlPlaneAdapter", () => { ok: true, }); } + const support = resolveStandardPrometheusSupportRoutes(url); + if (support !== undefined) { + return support; + } return Promise.reject(new Error(`unexpected fetch url: ${url}`)); }); - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - - gatewayListApi.mockResolvedValueOnce( - gatewayList( - [ - gateway({ phase: "Provisioning", status: "route pending" }), - gateway({ phase: "Failed", status: "apply error" }), - ], - 2, - 1, - ), - ); - const metrics = await adapter.getOperationalMetrics(context); - const provisionTimeMetric = metrics.metrics.find( - (metric) => metric.id === "provision-time", - ); - const memoryMetric = metrics.metrics.find( - (metric) => metric.id === "memory", - ); - expect(provisionTimeMetric).toBeUndefined(); - expect(memoryMetric).toEqual({ - id: "memory", - total: "1", - unit: "GiB", - value: "1", - }); + expect(metrics.failedSources).toEqual(["gateway-metrics"]); expect( - metrics.metrics.find((metric) => metric.id === "provisioned-gateways"), - ).toBeDefined(); - }); - - it("omits gateway-derived metrics for inconsistent pagination responses", async () => { - mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2); - - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - gatewayListApi.mockResolvedValueOnce(gatewayList([gateway()], 1, 2)); - - const metrics = await adapter.getOperationalMetrics(context); - - expect(metrics.failedSources).toEqual(["gateway-list"]); + metrics.metrics.find((metric) => metric.id === "provisioned-sandboxes"), + ).toBeUndefined(); expect( metrics.metrics.find((metric) => metric.id === "provisioned-gateways"), ).toBeUndefined(); @@ -619,32 +501,15 @@ describe("createDashboardControlPlaneAdapter", () => { ).toBeDefined(); }); - it("forwards abort signals to the gateway list client", async () => { + it("forwards abort signals to Prometheus metric fetches", async () => { const controller = new AbortController(); mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2); - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - gatewayListApi.mockResolvedValueOnce(gatewayList([gateway()], 1, 1)); - await adapter.getOperationalMetrics({ ...context, signal: controller.signal, }); - expect(gatewayListApi).toHaveBeenCalledWith( - { orderBy: "name asc", page: 1, size: 100 }, - { signal: controller.signal }, - ); - expect(usersListApi).toHaveBeenCalledWith( - { orderBy: "username asc", page: 1, size: 1 }, - { signal: controller.signal }, - ); expect(fetchMock).toHaveBeenCalledWith("/api/metrics/cluster-memory", { credentials: "same-origin", signal: controller.signal, @@ -661,10 +526,32 @@ describe("createDashboardControlPlaneAdapter", () => { credentials: "same-origin", signal: controller.signal, }); + expect(fetchMock).toHaveBeenCalledWith("/api/metrics/gateways", { + credentials: "same-origin", + signal: controller.signal, + }); + expect(fetchMock).toHaveBeenCalledWith("/api/metrics/gateway-sandboxes", { + credentials: "same-origin", + signal: controller.signal, + }); + expect(fetchMock).toHaveBeenCalledWith("/api/metrics/registered-users", { + credentials: "same-origin", + signal: controller.signal, + }); + expect(fetchMock).toHaveBeenCalledWith("/api/metrics/platform-inventory", { + credentials: "same-origin", + signal: controller.signal, + }); }); it("omits memory metrics when cluster memory is unavailable", async () => { fetchMock.mockImplementation((url: string) => { + if (url === "/api/metrics/gateways") { + return Promise.resolve({ + json: () => Promise.resolve({ counts: defaultGatewayPhaseCounts }), + ok: true, + }); + } if (url === "/api/metrics/cluster-memory") { return Promise.resolve({ ok: false, @@ -702,23 +589,12 @@ describe("createDashboardControlPlaneAdapter", () => { ok: true, }); } - if (url === "/api/metrics/gateway-provision-duration") { - return Promise.resolve({ - json: () => Promise.resolve(mockGatewayProvisionDurationResponse), - ok: true, - }); + const support = resolveStandardPrometheusSupportRoutes(url); + if (support !== undefined) { + return support; } return Promise.reject(new Error(`unexpected fetch url: ${url}`)); }); - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - gatewayListApi.mockResolvedValueOnce(gatewayList([gateway()], 1, 1)); - const metrics = await adapter.getOperationalMetrics(context); expect(metrics.failedSources).toEqual(["cluster-memory"]); @@ -732,6 +608,12 @@ describe("createDashboardControlPlaneAdapter", () => { it("omits CPU metrics when cluster CPU is unavailable", async () => { fetchMock.mockImplementation((url: string) => { + if (url === "/api/metrics/gateways") { + return Promise.resolve({ + json: () => Promise.resolve({ counts: defaultGatewayPhaseCounts }), + ok: true, + }); + } if (url === "/api/metrics/cluster-memory") { return Promise.resolve({ json: () => @@ -769,23 +651,12 @@ describe("createDashboardControlPlaneAdapter", () => { ok: true, }); } - if (url === "/api/metrics/gateway-provision-duration") { - return Promise.resolve({ - json: () => Promise.resolve(mockGatewayProvisionDurationResponse), - ok: true, - }); + const support = resolveStandardPrometheusSupportRoutes(url); + if (support !== undefined) { + return support; } return Promise.reject(new Error(`unexpected fetch url: ${url}`)); }); - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - gatewayListApi.mockResolvedValueOnce(gatewayList([gateway()], 1, 1)); - const metrics = await adapter.getOperationalMetrics(context); expect(metrics.failedSources).toEqual(["cluster-cpu"]); @@ -799,6 +670,12 @@ describe("createDashboardControlPlaneAdapter", () => { it("omits pod metrics when cluster pods are unavailable", async () => { fetchMock.mockImplementation((url: string) => { + if (url === "/api/metrics/gateways") { + return Promise.resolve({ + json: () => Promise.resolve({ counts: defaultGatewayPhaseCounts }), + ok: true, + }); + } if (url === "/api/metrics/cluster-memory") { return Promise.resolve({ json: () => @@ -838,23 +715,12 @@ describe("createDashboardControlPlaneAdapter", () => { ok: true, }); } - if (url === "/api/metrics/gateway-provision-duration") { - return Promise.resolve({ - json: () => Promise.resolve(mockGatewayProvisionDurationResponse), - ok: true, - }); + const support = resolveStandardPrometheusSupportRoutes(url); + if (support !== undefined) { + return support; } return Promise.reject(new Error(`unexpected fetch url: ${url}`)); }); - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - gatewayListApi.mockResolvedValueOnce(gatewayList([gateway()], 1, 1)); - const metrics = await adapter.getOperationalMetrics(context); expect(metrics.failedSources).toEqual(["cluster-pods"]); @@ -868,6 +734,12 @@ describe("createDashboardControlPlaneAdapter", () => { it("omits node metrics when cluster nodes are unavailable", async () => { fetchMock.mockImplementation((url: string) => { + if (url === "/api/metrics/gateways") { + return Promise.resolve({ + json: () => Promise.resolve({ counts: defaultGatewayPhaseCounts }), + ok: true, + }); + } if (url === "/api/metrics/cluster-memory") { return Promise.resolve({ json: () => @@ -905,17 +777,12 @@ describe("createDashboardControlPlaneAdapter", () => { status: 502, }); } + const support = resolveStandardPrometheusSupportRoutes(url); + if (support !== undefined) { + return support; + } return Promise.reject(new Error(`unexpected fetch url: ${url}`)); }); - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - gatewayListApi.mockResolvedValueOnce(gatewayList([gateway()], 1, 1)); - const metrics = await adapter.getOperationalMetrics(context); expect(metrics.failedSources).toEqual(["cluster-nodes"]); @@ -929,14 +796,6 @@ describe("createDashboardControlPlaneAdapter", () => { it("fails when every metric source is unavailable", async () => { fetchMock.mockRejectedValue(new Error("network down")); - usersListApi.mockRejectedValueOnce(new Error("users unavailable")); - gatewayListApi.mockRejectedValueOnce(new Error("gateways unavailable")); - managedClustersListApi.mockRejectedValueOnce( - new Error("managed clusters unavailable"), - ); - managedDatabasesListApi.mockRejectedValueOnce( - new Error("managed databases unavailable"), - ); await expect(adapter.getOperationalMetrics(context)).rejects.toThrow( "All operational dashboard metric sources failed", @@ -944,63 +803,47 @@ describe("createDashboardControlPlaneAdapter", () => { }); it("aggregates managed cluster and database inventory into operational metrics", async () => { - vi.useFakeTimers(); - vi.setSystemTime(new Date("2026-09-01T12:00:00.000Z")); - mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2); - - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - gatewayListApi.mockResolvedValueOnce(gatewayList([], 0, 1)); - - const firstClusterPage = Array.from({ length: 100 }, (_, index) => - managedCluster({ - created_at: - index < 2 ? "2026-08-15T10:00:00.000Z" : "2026-01-01T10:00:00.000Z", - id: `cluster-${String(index)}`, - name: `cluster-${String(index)}`, - provider: index < 5 ? "aws" : index < 7 ? "gcp" : "ibm", - region: index < 4 ? "us-east-1" : "eu-west-1", - status: index === 0 ? undefined : "Ready", - }), - ); - const secondClusterPage = Array.from({ length: 50 }, (_, index) => - managedCluster({ - created_at: "2026-01-01T10:00:00.000Z", - id: `cluster-${String(index + 100)}`, - name: `cluster-${String(index + 100)}`, - provider: "openshift", - region: " ", - status: "Failed", - }), - ); - - managedClustersListApi - .mockResolvedValueOnce(managedClusterList(firstClusterPage, 150, 1)) - .mockResolvedValueOnce(managedClusterList(secondClusterPage, 150, 2)); - managedDatabasesListApi.mockResolvedValueOnce( - managedDatabaseList( - [ - managedDatabase({ status: "Ready" }), - managedDatabase({ status: undefined }), - ], - 2, - 1, - ), + mockClusterMetricsResponses( + 1024 ** 3, + 512 * 1024 ** 2, + defaultGatewayPhaseCounts, + { + platformInventory: { + managed_clusters: { + by_provider: { + aws: 5, + gcp: 2, + ibm: 93, + openshift: 50, + }, + by_region: { + "eu-west-1 (aws)": 1, + "eu-west-1 (gcp)": 2, + "eu-west-1 (ibm)": 93, + "unknown (openshift)": 50, + "us-east-1 (aws)": 4, + }, + by_status: { + Failed: 50, + Ready: 99, + unknown: 1, + }, + created_last_30_days: 2, + total: 150, + }, + managed_databases: { + by_status: { + Ready: 1, + unknown: 1, + }, + total: 2, + }, + }, + }, ); const metrics = await adapter.getOperationalMetrics(context); - expect(managedClustersListApi).toHaveBeenCalledTimes(2); - expect(managedDatabasesListApi).toHaveBeenCalledWith( - { orderBy: "name asc", page: 1, size: 100 }, - { signal: undefined }, - ); - const clustersMetric = metrics.metrics.find( (metric) => metric.id === "managed-clusters", ); @@ -1039,25 +882,65 @@ describe("createDashboardControlPlaneAdapter", () => { }, value: "2", }); - - vi.useRealTimers(); }); - it("omits platform inventory metrics when managed database pagination is inconsistent", async () => { + it("omits platform inventory metrics when the platform-inventory route fails", async () => { mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2); - - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, + fetchMock.mockImplementation((url: string) => { + if (url === "/api/metrics/platform-inventory") { + return Promise.resolve({ ok: false, status: 502 }); + } + if (url === "/api/metrics/gateways") { + return Promise.resolve({ + json: () => Promise.resolve({ counts: defaultGatewayPhaseCounts }), + ok: true, + }); + } + if (url === "/api/metrics/cluster-memory") { + return Promise.resolve({ + json: () => + Promise.resolve({ + available_bytes: 512 * 1024 ** 2, + capacity_bytes: 1024 ** 3, + used_bytes: 512 * 1024 ** 2, + }), + ok: true, + }); + } + if (url === "/api/metrics/cluster-cpu") { + return Promise.resolve({ + json: () => + Promise.resolve({ + available_cores: 11.8, + capacity_cores: 60, + used_cores: 48.2, + }), + ok: true, + }); + } + if (url === "/api/metrics/cluster-pods") { + return Promise.resolve({ + json: () => Promise.resolve(mockClusterPodsResponse), + ok: true, + }); + } + if (url === "/api/metrics/cluster-nodes") { + return Promise.resolve({ + json: () => + Promise.resolve({ + not_ready_nodes: 0, + ready_nodes: 8, + total_nodes: 8, + }), + ok: true, + }); + } + const support = resolveStandardPrometheusSupportRoutes(url); + if (support !== undefined) { + return support; + } + return Promise.reject(new Error(`unexpected fetch url: ${url}`)); }); - gatewayListApi.mockResolvedValueOnce(gatewayList([], 0, 1)); - managedClustersListApi.mockResolvedValueOnce(managedClusterList([], 0, 1)); - managedDatabasesListApi.mockResolvedValueOnce( - managedDatabaseList([managedDatabase()], 1, 2), - ); const metrics = await adapter.getOperationalMetrics(context); @@ -1073,35 +956,18 @@ describe("createDashboardControlPlaneAdapter", () => { ).toBeDefined(); }); - it("forwards abort signals to managed inventory list clients", async () => { + it("forwards abort signals to the platform-inventory route", async () => { const controller = new AbortController(); mockClusterMetricsResponses(1024 ** 3, 512 * 1024 ** 2); - usersListApi.mockResolvedValueOnce({ - items: [], - kind: "UserList", - page: 1, - size: 1, - total: 0, - }); - gatewayListApi.mockResolvedValueOnce(gatewayList([], 0, 1)); - managedClustersListApi.mockResolvedValueOnce(managedClusterList([], 0, 1)); - managedDatabasesListApi.mockResolvedValueOnce( - managedDatabaseList([], 0, 1), - ); - await adapter.getOperationalMetrics({ ...context, signal: controller.signal, }); - expect(managedClustersListApi).toHaveBeenCalledWith( - { orderBy: "name asc", page: 1, size: 100 }, - { signal: controller.signal }, - ); - expect(managedDatabasesListApi).toHaveBeenCalledWith( - { orderBy: "name asc", page: 1, size: 100 }, - { signal: controller.signal }, - ); + expect(fetchMock).toHaveBeenCalledWith("/api/metrics/platform-inventory", { + credentials: "same-origin", + signal: controller.signal, + }); }); }); diff --git a/components/web-console/app/adapters/api/dashboard-control-plane.ts b/components/web-console/app/adapters/api/dashboard-control-plane.ts index b129d3628..978b1d754 100644 --- a/components/web-console/app/adapters/api/dashboard-control-plane.ts +++ b/components/web-console/app/adapters/api/dashboard-control-plane.ts @@ -1,5 +1,6 @@ import { - aggregateGatewayDisplayStatusCounts, + fetchGatewayMetrics, + gatewayPhaseCountsToDisplayStatusCounts, type GatewayDisplayStatusCounts, } from "@openshift-online/hypershell-gateway-management-ui"; import type { @@ -12,15 +13,12 @@ import type { import type { SDKClient } from "@openshift-online/hypershell-sdk"; import { - aggregateManagedClusterList, - aggregateManagedDatabaseList, - buildManagedClustersMetric, - buildManagedDatabasesMetric, + platformInventoryMetricsResponseToMetrics, + type PlatformInventoryMetricsResponse, } from "./platform-inventory-aggregation"; type DashboardApiFactory = (correlationId: string) => SDKClient; -const gatewayListPageSize = 100; const gibibyteDivisor = 1024 ** 3; const secondsPerMinute = 60; @@ -60,6 +58,14 @@ interface GatewayProvisionDurationResponse { p95_seconds: number; } +interface GatewaySandboxesResponse { + active_sandboxes: number; +} + +interface RegisteredUsersResponse { + total_registered: number; +} + function bytesToRoundedGib(bytes: number): string { return String(Math.round(bytes / gibibyteDivisor)); } @@ -221,10 +227,25 @@ function gatewayDisplayCountsToMetric( }; } -interface GatewayListAggregate { - activeSandboxCount: number; - displayStatusCounts: GatewayDisplayStatusCounts; - total: number; +async function fetchGatewaySandboxesMetric( + signal?: AbortSignal, +): Promise { + const response = await fetch("/api/metrics/gateway-sandboxes", { + credentials: "same-origin", + signal, + }); + if (!response.ok) { + throw new Error( + `Failed to fetch gateway sandbox metrics: ${String(response.status)}`, + ); + } + + const body = (await response.json()) as GatewaySandboxesResponse; + + return { + id: "provisioned-sandboxes", + value: String(body.active_sandboxes), + }; } function isAbortError(error: unknown): boolean { @@ -236,76 +257,29 @@ function isAbortError(error: unknown): boolean { ); } -async function aggregateGatewayList( - context: DashboardInvocationContext, - apiFactory: DashboardApiFactory, -): Promise { - const client = apiFactory(context.correlationId); - let page = 1; - let total = 0; - let activeSandboxCount = 0; - const lifecycleRecords: { phase?: string; status?: string }[] = []; - - do { - const result = await client.gateways.list( - { - orderBy: "name asc", - page, - size: gatewayListPageSize, - }, - { signal: context.signal }, - ); - - if ( - result.page !== page || - result.total < 0 || - result.items.length > - Math.max( - 0, - Math.min( - gatewayListPageSize, - result.total - (page - 1) * gatewayListPageSize, - ), - ) - ) { - throw new Error("Gateway list response was inconsistent"); - } - - for (const gateway of result.items) { - const sandboxCount = gateway.active_sandbox_count; - activeSandboxCount += typeof sandboxCount === "number" ? sandboxCount : 0; - lifecycleRecords.push({ - phase: gateway.phase, - status: gateway.status, - }); - } - - total = result.total; - page += 1; - } while ((page - 1) * gatewayListPageSize < total); +async function fetchGatewayPrometheusMetric( + signal?: AbortSignal, +): Promise { + const phaseCounts = await fetchGatewayMetrics(signal); + const displayStatusCounts = + gatewayPhaseCountsToDisplayStatusCounts(phaseCounts); + const total = Object.values(phaseCounts).reduce( + (sum, count) => sum + count, + 0, + ); - return { - activeSandboxCount, - displayStatusCounts: aggregateGatewayDisplayStatusCounts(lifecycleRecords), - total, - }; + return gatewayDisplayCountsToMetric(total, displayStatusCounts); } -async function fetchGatewayListMetrics( +async function fetchGatewayPrometheusMetrics( context: DashboardInvocationContext, - apiFactory: DashboardApiFactory, ): Promise { - const aggregate = await aggregateGatewayList(context, apiFactory); - const metrics: OperationalMetric[] = [ - gatewayDisplayCountsToMetric( - aggregate.total, - aggregate.displayStatusCounts, - ), - { - id: "provisioned-sandboxes", - value: String(aggregate.activeSandboxCount), - }, - ]; + const [gatewayMetric, sandboxMetric] = await Promise.all([ + fetchGatewayPrometheusMetric(context.signal), + fetchGatewaySandboxesMetric(context.signal), + ]); + + const metrics: OperationalMetric[] = [gatewayMetric, sandboxMetric]; const provisionTimeMetric = await fetchGatewayProvisionDurationMetric( context.signal, @@ -319,36 +293,42 @@ async function fetchGatewayListMetrics( async function fetchRegisteredUsersMetric( context: DashboardInvocationContext, - apiFactory: DashboardApiFactory, ): Promise { - const client = apiFactory(context.correlationId); - const userList = await client.users.list( - { orderBy: "username asc", page: 1, size: 1 }, - { signal: context.signal }, - ); + const response = await fetch("/api/metrics/registered-users", { + credentials: "same-origin", + signal: context.signal, + }); + if (!response.ok) { + throw new Error( + `Failed to fetch registered user metrics: ${String(response.status)}`, + ); + } + + const body = (await response.json()) as RegisteredUsersResponse; return [ { id: "registered-users", - value: String(userList.total), + value: String(body.total_registered), }, ]; } async function fetchPlatformInventoryMetrics( context: DashboardInvocationContext, - apiFactory: DashboardApiFactory, ): Promise { - const client = apiFactory(context.correlationId); - const [clusterAggregate, databaseAggregate] = await Promise.all([ - aggregateManagedClusterList(client, context.signal), - aggregateManagedDatabaseList(client, context.signal), - ]); + const response = await fetch("/api/metrics/platform-inventory", { + credentials: "same-origin", + signal: context.signal, + }); + if (!response.ok) { + throw new Error( + `Failed to fetch platform inventory metrics: ${String(response.status)}`, + ); + } - return [ - buildManagedClustersMetric(clusterAggregate), - buildManagedDatabasesMetric(databaseAggregate), - ]; + const body = (await response.json()) as PlatformInventoryMetricsResponse; + return platformInventoryMetricsResponseToMetrics(body); } interface MetricSourceDefinition { @@ -361,16 +341,16 @@ interface MetricSourceDefinition { const metricSources: readonly MetricSourceDefinition[] = [ { - id: "gateway-list", - fetch: fetchGatewayListMetrics, + id: "gateway-metrics", + fetch: async (context) => fetchGatewayPrometheusMetrics(context), }, { id: "registered-users", - fetch: fetchRegisteredUsersMetric, + fetch: async (context) => fetchRegisteredUsersMetric(context), }, { id: "platform-inventory", - fetch: fetchPlatformInventoryMetrics, + fetch: async (context) => fetchPlatformInventoryMetrics(context), }, { id: "cluster-memory", diff --git a/components/web-console/app/adapters/api/platform-inventory-aggregation.ts b/components/web-console/app/adapters/api/platform-inventory-aggregation.ts index 09b55b2f1..5943025f8 100644 --- a/components/web-console/app/adapters/api/platform-inventory-aggregation.ts +++ b/components/web-console/app/adapters/api/platform-inventory-aggregation.ts @@ -209,3 +209,37 @@ export function buildManagedDatabasesMetric( value: String(aggregate.total), }; } + +export interface PlatformInventoryMetricsResponse { + managed_clusters: { + by_provider: Record; + by_region: Record; + by_status: Record; + created_last_30_days: number; + total: number; + }; + managed_databases: { + by_status: Record; + total: number; + }; +} + +export function platformInventoryMetricsResponseToMetrics( + response: PlatformInventoryMetricsResponse, +): OperationalMetric[] { + return [ + { + createdLast30Days: String(response.managed_clusters.created_last_30_days), + id: "managed-clusters", + inventoryProviders: response.managed_clusters.by_provider, + inventoryRegions: response.managed_clusters.by_region, + inventoryStatus: response.managed_clusters.by_status, + value: String(response.managed_clusters.total), + }, + { + id: "managed-databases", + inventoryStatus: response.managed_databases.by_status, + value: String(response.managed_databases.total), + }, + ]; +} diff --git a/components/web-console/app/features/dashboard/require-dashboard-admin.test.tsx b/components/web-console/app/features/dashboard/require-dashboard-admin.test.tsx index cea9cc7ab..b01118326 100644 --- a/components/web-console/app/features/dashboard/require-dashboard-admin.test.tsx +++ b/components/web-console/app/features/dashboard/require-dashboard-admin.test.tsx @@ -63,7 +63,7 @@ describe("RequireDashboardAdmin", () => { expect(screen.queryByTestId("dashboard-content")).toBeNull(); }); - it("renders children for hypershell-admins", async () => { + it("shows access denied for hypershell-admins without platform:admin", async () => { getSessionMock.mockResolvedValue({ authenticated: true, authEnabled: true, @@ -72,7 +72,10 @@ describe("RequireDashboardAdmin", () => { renderGuard(); - expect(await screen.findByTestId("dashboard-content")).toBeTruthy(); + expect( + await screen.findByRole("heading", { name: "Access denied" }), + ).toBeTruthy(); + expect(screen.queryByTestId("dashboard-content")).toBeNull(); }); it("renders children for platform:admin", async () => { diff --git a/components/web-console/app/lib/session-roles.test.ts b/components/web-console/app/lib/session-roles.test.ts index 7b5f728d2..3b745ede4 100644 --- a/components/web-console/app/lib/session-roles.test.ts +++ b/components/web-console/app/lib/session-roles.test.ts @@ -3,10 +3,10 @@ import { describe, expect, it } from "vitest"; import { hasDashboardAdminRole } from "./session-roles"; describe("hasDashboardAdminRole", () => { - it("returns true when hypershell-admins is present", () => { + it("returns false when only hypershell-admins is present", () => { expect( hasDashboardAdminRole(["hypershell-users", "hypershell-admins"]), - ).toBe(true); + ).toBe(false); }); it("returns true when platform:admin is present", () => { diff --git a/components/web-console/bff/src/app.ts b/components/web-console/bff/src/app.ts index 7d8070f5d..916686766 100644 --- a/components/web-console/bff/src/app.ts +++ b/components/web-console/bff/src/app.ts @@ -23,6 +23,9 @@ import { queryGatewayPhaseCounts } from "./metrics-gateways.js"; import { queryClusterCpu } from "./metrics-cluster-cpu.js"; import { queryClusterMemory } from "./metrics-cluster-memory.js"; import { queryGatewayProvisionDuration } from "./metrics-gateway-provision-duration.js"; +import { queryGatewaySandboxes } from "./metrics-gateway-sandboxes.js"; +import { queryPlatformInventory } from "./metrics-platform-inventory.js"; +import { queryRegisteredUsers } from "./metrics-registered-users.js"; import { queryClusterPods } from "./metrics-cluster-pods.js"; import { queryClusterNodes } from "./metrics-cluster-nodes.js"; import { tokenExpired } from "./tokens.js"; @@ -581,6 +584,69 @@ export async function buildApp( }, ); + app.get( + "/api/metrics/gateway-sandboxes", + { preHandler: requireDashboardMetricsAccess }, + async (request, reply) => { + try { + return await queryGatewaySandboxes( + config.prometheusUrl, + config.prometheusQueryTimeoutMs, + config.prometheusNamespace, + ); + } catch (error) { + request.log.warn( + { err: error }, + "gateway sandbox metrics query failed", + ); + reply.code(502); + return { error: "Metrics unavailable", statusCode: 502 }; + } + }, + ); + + app.get( + "/api/metrics/platform-inventory", + { preHandler: requireDashboardMetricsAccess }, + async (request, reply) => { + try { + return await queryPlatformInventory( + config.prometheusUrl, + config.prometheusQueryTimeoutMs, + config.prometheusNamespace, + ); + } catch (error) { + request.log.warn( + { err: error }, + "platform inventory metrics query failed", + ); + reply.code(502); + return { error: "Metrics unavailable", statusCode: 502 }; + } + }, + ); + + app.get( + "/api/metrics/registered-users", + { preHandler: requireDashboardMetricsAccess }, + async (request, reply) => { + try { + return await queryRegisteredUsers( + config.prometheusUrl, + config.prometheusQueryTimeoutMs, + config.prometheusNamespace, + ); + } catch (error) { + request.log.warn( + { err: error }, + "registered user metrics query failed", + ); + reply.code(502); + return { error: "Metrics unavailable", statusCode: 502 }; + } + }, + ); + app.all("/api/*", async (request, reply) => { // Start one BFF server span per proxied request. It continues a valid // inbound W3C context and yields the validated upstream context to set on diff --git a/components/web-console/bff/src/metrics-gateway-sandboxes.ts b/components/web-console/bff/src/metrics-gateway-sandboxes.ts new file mode 100644 index 000000000..2f41537d2 --- /dev/null +++ b/components/web-console/bff/src/metrics-gateway-sandboxes.ts @@ -0,0 +1,26 @@ +import { + applicationScalarQuery, + queryPrometheusInstantScalar, +} from "./prometheus-instant-query.js"; +import type { MetricsSource } from "./metrics-source.js"; + +export const gatewayActiveSandboxesPromql = + "hypershell_gateways_active_sandboxes_total"; + +export interface GatewaySandboxesCounts { + active_sandboxes: number; +} + +export async function queryGatewaySandboxes( + prometheusUrl: MetricsSource, + timeoutMs: number, + namespace?: string, +): Promise { + const activeSandboxes = await queryPrometheusInstantScalar( + prometheusUrl, + applicationScalarQuery(gatewayActiveSandboxesPromql, namespace), + timeoutMs, + ); + + return { active_sandboxes: activeSandboxes }; +} diff --git a/components/web-console/bff/src/metrics-platform-inventory.ts b/components/web-console/bff/src/metrics-platform-inventory.ts new file mode 100644 index 000000000..274a99c72 --- /dev/null +++ b/components/web-console/bff/src/metrics-platform-inventory.ts @@ -0,0 +1,134 @@ +import type { MetricsSource } from "./metrics-source.js"; +import { + applicationScalarQuery, + applicationVectorQuery, + queryPrometheusInstantScalar, + queryPrometheusInstantVector, +} from "./prometheus-instant-query.js"; + +export const managedClustersTotalPromql = "hypershell_managed_clusters_total"; +export const managedClustersCreatedLast30DaysPromql = + "hypershell_managed_clusters_created_last_30_days_total"; +export const managedClustersInventoryPromql = + "hypershell_managed_clusters_inventory_total"; +export const managedDatabasesTotalPromql = "hypershell_managed_databases_total"; +export const managedDatabasesInventoryPromql = + "hypershell_managed_databases_inventory_total"; + +const clusterInventoryGroupBy = ["status", "provider", "region"] as const; +const databaseInventoryGroupBy = ["status"] as const; + +export interface ManagedClustersInventoryResponse { + by_provider: Record; + by_region: Record; + by_status: Record; + created_last_30_days: number; + total: number; +} + +export interface ManagedDatabasesInventoryResponse { + by_status: Record; + total: number; +} + +export interface PlatformInventoryResponse { + managed_clusters: ManagedClustersInventoryResponse; + managed_databases: ManagedDatabasesInventoryResponse; +} + +function incrementBucket( + buckets: Record, + key: string, + amount: number, +): void { + buckets[key] = (buckets[key] ?? 0) + amount; +} + +function formatPlacementLabel(region: string, provider: string): string { + return `${region} (${provider})`; +} + +export async function queryPlatformInventory( + prometheusUrl: MetricsSource, + timeoutMs: number, + namespace?: string, +): Promise { + const [ + clusterTotal, + clustersCreatedLast30Days, + clusterInventory, + databaseTotal, + databaseInventory, + ] = await Promise.all([ + queryPrometheusInstantScalar( + prometheusUrl, + applicationScalarQuery(managedClustersTotalPromql, namespace), + timeoutMs, + ), + queryPrometheusInstantScalar( + prometheusUrl, + applicationScalarQuery(managedClustersCreatedLast30DaysPromql, namespace), + timeoutMs, + ), + queryPrometheusInstantVector( + prometheusUrl, + applicationVectorQuery( + managedClustersInventoryPromql, + clusterInventoryGroupBy, + namespace, + ), + timeoutMs, + ), + queryPrometheusInstantScalar( + prometheusUrl, + applicationScalarQuery(managedDatabasesTotalPromql, namespace), + timeoutMs, + ), + queryPrometheusInstantVector( + prometheusUrl, + applicationVectorQuery( + managedDatabasesInventoryPromql, + databaseInventoryGroupBy, + namespace, + ), + timeoutMs, + ), + ]); + + const byStatus: Record = {}; + const byProvider: Record = {}; + const byRegion: Record = {}; + + for (const sample of clusterInventory) { + const status = sample.labels.status ?? "unknown"; + const provider = sample.labels.provider ?? "unknown"; + const region = sample.labels.region ?? "unknown"; + incrementBucket(byStatus, status, sample.value); + incrementBucket(byProvider, provider, sample.value); + incrementBucket( + byRegion, + formatPlacementLabel(region, provider), + sample.value, + ); + } + + const databaseByStatus: Record = {}; + for (const sample of databaseInventory) { + const status = sample.labels.status ?? "unknown"; + incrementBucket(databaseByStatus, status, sample.value); + } + + return { + managed_clusters: { + by_provider: byProvider, + by_region: byRegion, + by_status: byStatus, + created_last_30_days: clustersCreatedLast30Days, + total: clusterTotal, + }, + managed_databases: { + by_status: databaseByStatus, + total: databaseTotal, + }, + }; +} diff --git a/components/web-console/bff/src/metrics-registered-users.ts b/components/web-console/bff/src/metrics-registered-users.ts new file mode 100644 index 000000000..8a72cc749 --- /dev/null +++ b/components/web-console/bff/src/metrics-registered-users.ts @@ -0,0 +1,25 @@ +import { + applicationScalarQuery, + queryPrometheusInstantScalar, +} from "./prometheus-instant-query.js"; +import type { MetricsSource } from "./metrics-source.js"; + +export const usersRegisteredTotalPromql = "hypershell_users_registered_total"; + +export interface RegisteredUsersResponse { + total_registered: number; +} + +export async function queryRegisteredUsers( + prometheusUrl: MetricsSource, + timeoutMs: number, + namespace?: string, +): Promise { + const totalRegistered = await queryPrometheusInstantScalar( + prometheusUrl, + applicationScalarQuery(usersRegisteredTotalPromql, namespace), + timeoutMs, + ); + + return { total_registered: totalRegistered }; +} diff --git a/components/web-console/bff/src/prometheus-instant-query.ts b/components/web-console/bff/src/prometheus-instant-query.ts new file mode 100644 index 000000000..95c6aea97 --- /dev/null +++ b/components/web-console/bff/src/prometheus-instant-query.ts @@ -0,0 +1,129 @@ +import { + fetchMetrics, + namespaceSelector, + type MetricsSource, +} from "./metrics-source.js"; + +interface PrometheusQueryResponse { + status: string; + data?: { + result: { + metric: Record; + value: [string, string]; + }[]; + }; +} + +export interface PrometheusLabeledSample { + labels: Record; + value: number; +} + +/** Build a scalar application-metric query, deduplicating API replica series. */ +export function applicationScalarQuery( + metric: string, + namespace?: string, +): string { + if (!namespace) { + return metric; + } + return `max(${metric}${namespaceSelector(namespace)})`; +} + +/** Build a labeled application-metric query, deduplicating API replica series. */ +export function applicationVectorQuery( + metric: string, + groupBy: readonly string[], + namespace?: string, +): string { + if (!namespace) { + return metric; + } + const labels = groupBy.join(", "); + return `max by (${labels}) (${metric}${namespaceSelector(namespace)})`; +} + +async function queryPrometheus( + source: MetricsSource, + query: string, + timeoutMs: number, +): Promise { + const controller = new AbortController(); + const timeoutReason = new Error("Prometheus query timed out"); + const timeout = setTimeout(() => { + controller.abort(timeoutReason); + }, timeoutMs); + + try { + const response = await fetchMetrics(source, query, controller.signal); + if (!response.ok) { + throw new Error("Prometheus query request failed"); + } + + const body = (await response.json()) as PrometheusQueryResponse; + if (body.status !== "success") { + throw new Error("Prometheus query returned non-success status"); + } + + return body; + } finally { + clearTimeout(timeout); + } +} + +function parseSampleValue(rawValue: string | undefined): number { + if (rawValue === undefined) { + throw new Error("Prometheus query returned invalid sample"); + } + + const value = Number(rawValue); + if (!Number.isFinite(value) || value < 0) { + throw new Error("Prometheus query returned invalid sample"); + } + + return value; +} + +export async function queryPrometheusInstantScalar( + source: MetricsSource, + query: string, + timeoutMs: number, +): Promise { + const body = await queryPrometheus(source, query, timeoutMs); + const samples = body.data?.result ?? []; + if (samples.length === 0) { + throw new Error("Prometheus query returned no samples"); + } + + let maxValue = 0; + for (const sample of samples) { + maxValue = Math.max( + maxValue, + Math.round(parseSampleValue(sample.value[1])), + ); + } + + return maxValue; +} + +export async function queryPrometheusInstantVector( + source: MetricsSource, + query: string, + timeoutMs: number, +): Promise { + const body = await queryPrometheus(source, query, timeoutMs); + + const samplesByLabels = new Map(); + for (const sample of body.data?.result ?? []) { + const value = Math.round(parseSampleValue(sample.value[1])); + const key = JSON.stringify(sample.metric); + const existing = samplesByLabels.get(key); + if (existing === undefined) { + samplesByLabels.set(key, { labels: sample.metric, value }); + continue; + } + existing.value = Math.max(existing.value, value); + } + + return [...samplesByLabels.values()]; +} diff --git a/components/web-console/bff/src/roles.test.ts b/components/web-console/bff/src/roles.test.ts index 0e900ca99..78dcafb61 100644 --- a/components/web-console/bff/src/roles.test.ts +++ b/components/web-console/bff/src/roles.test.ts @@ -3,10 +3,10 @@ import { describe, expect, it } from "vitest"; import { hasDashboardAdminRole } from "./roles.js"; describe("hasDashboardAdminRole", () => { - it("returns true when hypershell-admins is present", () => { + it("returns false when only hypershell-admins is present", () => { expect( hasDashboardAdminRole(["hypershell-users", "hypershell-admins"]), - ).toBe(true); + ).toBe(false); }); it("returns true when platform:admin is present", () => { diff --git a/components/web-console/bff/test/auth.test.ts b/components/web-console/bff/test/auth.test.ts index a0ca9ad2f..71d485e6e 100644 --- a/components/web-console/bff/test/auth.test.ts +++ b/components/web-console/bff/test/auth.test.ts @@ -847,6 +847,9 @@ describe("web-console BFF with OIDC enabled", () => { "/api/metrics/cluster-pods", "/api/metrics/cluster-nodes", "/api/metrics/gateway-provision-duration", + "/api/metrics/gateway-sandboxes", + "/api/metrics/platform-inventory", + "/api/metrics/registered-users", ]) { const response = await app.inject({ headers: { cookie }, @@ -862,7 +865,7 @@ describe("web-console BFF with OIDC enabled", () => { } }); - it("serves /dashboard to hypershell-admins", async () => { + it("redirects hypershell-admins away from /dashboard", async () => { const session = app.createSecureSession({ accessToken: "test-access-token", expiresAt: Math.floor(Date.now() / 1000) + 3600, @@ -875,8 +878,8 @@ describe("web-console BFF with OIDC enabled", () => { url: "/dashboard", }); - expect(response.statusCode).toBe(200); - expect(response.headers["content-type"]).toContain("text/html"); + expect(response.statusCode).toBe(302); + expect(response.headers.location).toBe("/"); }); it("serves /dashboard to platform:admin", async () => { @@ -896,7 +899,7 @@ describe("web-console BFF with OIDC enabled", () => { expect(response.headers["content-type"]).toContain("text/html"); }); - it("serves /metrics to hypershell-admins", async () => { + it("redirects hypershell-admins away from /metrics", async () => { const session = app.createSecureSession({ accessToken: "test-access-token", expiresAt: Math.floor(Date.now() / 1000) + 3600, @@ -909,8 +912,8 @@ describe("web-console BFF with OIDC enabled", () => { url: "/metrics", }); - expect(response.statusCode).toBe(200); - expect(response.headers["content-type"]).toContain("text/html"); + expect(response.statusCode).toBe(302); + expect(response.headers.location).toBe("/"); }); it("serves application routes when authenticated", async () => { diff --git a/components/web-console/bff/test/metrics-cluster-cpu-route.test.ts b/components/web-console/bff/test/metrics-cluster-cpu-route.test.ts index 60ed59d47..841d0be2d 100644 --- a/components/web-console/bff/test/metrics-cluster-cpu-route.test.ts +++ b/components/web-console/bff/test/metrics-cluster-cpu-route.test.ts @@ -273,7 +273,7 @@ describe("GET /api/metrics/cluster-cpu", () => { expiresAt: Math.floor(Date.now() / 1000) + 3600, name: "Test User", preferredUsername: "testuser", - roles: ["hypershell-admins"], + roles: ["platform:admin"], sub: "user-123", }); diff --git a/components/web-console/bff/test/metrics-cluster-memory-route.test.ts b/components/web-console/bff/test/metrics-cluster-memory-route.test.ts index 20cc41742..641688b1c 100644 --- a/components/web-console/bff/test/metrics-cluster-memory-route.test.ts +++ b/components/web-console/bff/test/metrics-cluster-memory-route.test.ts @@ -273,7 +273,7 @@ describe("GET /api/metrics/cluster-memory", () => { expiresAt: Math.floor(Date.now() / 1000) + 3600, name: "Test User", preferredUsername: "testuser", - roles: ["hypershell-admins"], + roles: ["platform:admin"], sub: "user-123", }); diff --git a/components/web-console/bff/test/metrics-cluster-nodes-route.test.ts b/components/web-console/bff/test/metrics-cluster-nodes-route.test.ts index 8f304ae22..db5d76a4b 100644 --- a/components/web-console/bff/test/metrics-cluster-nodes-route.test.ts +++ b/components/web-console/bff/test/metrics-cluster-nodes-route.test.ts @@ -273,7 +273,7 @@ describe("GET /api/metrics/cluster-nodes", () => { expiresAt: Math.floor(Date.now() / 1000) + 3600, name: "Test User", preferredUsername: "testuser", - roles: ["hypershell-admins"], + roles: ["platform:admin"], sub: "user-123", }); diff --git a/components/web-console/bff/test/metrics-cluster-pods-route.test.ts b/components/web-console/bff/test/metrics-cluster-pods-route.test.ts index cc808363e..bbe239a70 100644 --- a/components/web-console/bff/test/metrics-cluster-pods-route.test.ts +++ b/components/web-console/bff/test/metrics-cluster-pods-route.test.ts @@ -323,7 +323,7 @@ describe("GET /api/metrics/cluster-pods", () => { expiresAt: Math.floor(Date.now() / 1000) + 3600, name: "Test User", preferredUsername: "testuser", - roles: ["hypershell-admins"], + roles: ["platform:admin"], sub: "user-123", }); diff --git a/components/web-console/bff/test/metrics-platform-inventory.test.ts b/components/web-console/bff/test/metrics-platform-inventory.test.ts new file mode 100644 index 000000000..c0f2a0d9d --- /dev/null +++ b/components/web-console/bff/test/metrics-platform-inventory.test.ts @@ -0,0 +1,224 @@ +import { + createServer, + type IncomingMessage, + type ServerResponse, +} from "node:http"; + +import { describe, expect, it } from "vitest"; + +import { queryPlatformInventory } from "../src/metrics-platform-inventory.js"; + +async function startPrometheusStub( + handler: (request: IncomingMessage, response: ServerResponse) => void, +): Promise<{ close: () => void; port: number }> { + const server = createServer(handler); + await new Promise((resolve) => { + server.listen(0, "127.0.0.1", resolve); + }); + const address = server.address(); + if (address === null || typeof address === "string") { + throw new Error("expected tcp listener address"); + } + return { + close: () => server.close(), + port: address.port, + }; +} + +function prometheusScalarSample(value: string) { + return JSON.stringify({ + status: "success", + data: { + result: [ + { + metric: {}, + value: ["1704067200", value], + }, + ], + }, + }); +} + +describe("queryPlatformInventory", () => { + it("aggregates labeled cluster and database inventory samples", async () => { + const prometheus = await startPrometheusStub((request, response) => { + const url = new URL(request.url ?? "/", "http://127.0.0.1"); + const query = url.searchParams.get("query") ?? ""; + response.setHeader("content-type", "application/json"); + + if (query === "hypershell_managed_clusters_total") { + response.end(prometheusScalarSample("150")); + return; + } + if (query === "hypershell_managed_clusters_created_last_30_days_total") { + response.end(prometheusScalarSample("2")); + return; + } + if (query === "hypershell_managed_databases_total") { + response.end(prometheusScalarSample("2")); + return; + } + if (query === "hypershell_managed_clusters_inventory_total") { + response.end( + JSON.stringify({ + status: "success", + data: { + result: [ + { + metric: { + provider: "aws", + region: "us-east-1", + status: "Ready", + }, + value: ["1704067200", "4"], + }, + { + metric: { + provider: "openshift", + region: "unknown", + status: "Failed", + }, + value: ["1704067200", "50"], + }, + ], + }, + }), + ); + return; + } + if (query === "hypershell_managed_databases_inventory_total") { + response.end( + JSON.stringify({ + status: "success", + data: { + result: [ + { + metric: { status: "Ready" }, + value: ["1704067200", "1"], + }, + { + metric: { status: "unknown" }, + value: ["1704067200", "1"], + }, + ], + }, + }), + ); + return; + } + + response.statusCode = 404; + response.end(); + }); + + const result = await queryPlatformInventory( + `http://127.0.0.1:${String(prometheus.port)}`, + 5_000, + ); + + expect(result.managed_clusters.total).toBe(150); + expect(result.managed_clusters.created_last_30_days).toBe(2); + expect(result.managed_clusters.by_status).toEqual({ + Failed: 50, + Ready: 4, + }); + expect(result.managed_databases.total).toBe(2); + expect(result.managed_databases.by_status).toEqual({ + Ready: 1, + unknown: 1, + }); + + prometheus.close(); + }); + + it("scopes application inventory queries to the configured namespace", async () => { + const queries: string[] = []; + const prometheus = await startPrometheusStub((request, response) => { + const query = + new URL(request.url ?? "/", "http://127.0.0.1").searchParams.get( + "query", + ) ?? ""; + queries.push(query); + response.setHeader("content-type", "application/json"); + if (query.includes("inventory_total")) { + response.end( + JSON.stringify({ status: "success", data: { result: [] } }), + ); + return; + } + response.end(prometheusScalarSample("0")); + }); + + await queryPlatformInventory( + `http://127.0.0.1:${String(prometheus.port)}`, + 5_000, + "hyp1", + ); + + expect(queries).toEqual([ + 'max(hypershell_managed_clusters_total{namespace="hyp1"})', + 'max(hypershell_managed_clusters_created_last_30_days_total{namespace="hyp1"})', + 'max by (status, provider, region) (hypershell_managed_clusters_inventory_total{namespace="hyp1"})', + 'max(hypershell_managed_databases_total{namespace="hyp1"})', + 'max by (status) (hypershell_managed_databases_inventory_total{namespace="hyp1"})', + ]); + + prometheus.close(); + }); + + it("does not report zero totals when scalar samples are missing", async () => { + const prometheus = await startPrometheusStub((_request, response) => { + response.setHeader("content-type", "application/json"); + response.end(JSON.stringify({ status: "success", data: { result: [] } })); + }); + + await expect( + queryPlatformInventory( + `http://127.0.0.1:${String(prometheus.port)}`, + 5_000, + ), + ).rejects.toThrow("Prometheus query returned no samples"); + + prometheus.close(); + }); + + it("keeps repeated gauge samples from adding or overwriting totals", async () => { + const prometheus = await startPrometheusStub((request, response) => { + const query = + new URL(request.url ?? "/", "http://127.0.0.1").searchParams.get( + "query", + ) ?? ""; + response.setHeader("content-type", "application/json"); + if (query === "hypershell_managed_clusters_total") { + response.end( + JSON.stringify({ + status: "success", + data: { + result: [3, 3, 2].map((value) => ({ + metric: {}, + value: ["1704067200", String(value)], + })), + }, + }), + ); + return; + } + if (query.includes("inventory_total")) { + response.end( + JSON.stringify({ status: "success", data: { result: [] } }), + ); + return; + } + response.end(prometheusScalarSample("0")); + }); + + const result = await queryPlatformInventory( + `http://127.0.0.1:${String(prometheus.port)}`, + 5_000, + ); + + expect(result.managed_clusters.total).toBe(3); + + prometheus.close(); + }); +}); diff --git a/components/web-console/shared/dashboard-roles.ts b/components/web-console/shared/dashboard-roles.ts index dbe38a252..73a1ff040 100644 --- a/components/web-console/shared/dashboard-roles.ts +++ b/components/web-console/shared/dashboard-roles.ts @@ -4,10 +4,7 @@ export const HYPERSHELL_ADMIN_ROLE = "hypershell-admins"; /** Keycloak realm role for platform-wide administration. */ export const PLATFORM_ADMIN_ROLE = "platform:admin"; -const DASHBOARD_ADMIN_ROLES = new Set([ - HYPERSHELL_ADMIN_ROLE, - PLATFORM_ADMIN_ROLE, -]); +const DASHBOARD_ADMIN_ROLES = new Set([PLATFORM_ADMIN_ROLE]); export function hasDashboardAdminRole(roles: readonly string[]): boolean { return roles.some((role) => DASHBOARD_ADMIN_ROLES.has(role)); diff --git a/deploy/base/keycloak/keycloak.yaml b/deploy/base/keycloak/keycloak.yaml index 174b30759..2a0308f7a 100644 --- a/deploy/base/keycloak/keycloak.yaml +++ b/deploy/base/keycloak/keycloak.yaml @@ -541,7 +541,7 @@ data: "emailVerified": true, "enabled": true, "credentials": [{ "type": "password", "value": "admin", "temporary": false }], - "realmRoles": ["hypershell-admins", "hypershell-users", "gateway:creator"] + "realmRoles": ["hypershell-admins", "hypershell-users", "gateway:creator", "platform:admin"] }, { "username": "developer", diff --git a/packages/gateway-management-ui/src/gateways/gateway-data.test.ts b/packages/gateway-management-ui/src/gateways/gateway-data.test.ts index ef5e9594b..b143ed562 100644 --- a/packages/gateway-management-ui/src/gateways/gateway-data.test.ts +++ b/packages/gateway-management-ui/src/gateways/gateway-data.test.ts @@ -4,6 +4,7 @@ import { normalizeGatewayPlacementClusterIds } from "../application/gateway-plac import type { GatewayRecord } from "../application/gateway-types"; import { aggregateGatewayDisplayStatusCounts, + gatewayPhaseCountsToDisplayStatusCounts, gatewayConsoleReadyDeadlineMilliseconds, gatewayConsoleUnavailable, gatewayNeedsStatusPolling, @@ -336,6 +337,23 @@ describe("gateway presentation data", () => { }); }); + it("maps Prometheus phase counts into dashboard status buckets", () => { + expect( + gatewayPhaseCountsToDisplayStatusCounts({ + Pending: 2, + Provisioning: 3, + Running: 10, + Degraded: 1, + Failed: 4, + }), + ).toEqual({ + degraded: 1, + failed: 4, + healthy: 10, + provisioning: 5, + }); + }); + it("keeps a returned cluster identifier for name resolution only", () => { expect( toGatewayConnection( diff --git a/packages/gateway-management-ui/src/gateways/gateway-data.ts b/packages/gateway-management-ui/src/gateways/gateway-data.ts index 5536b431d..a77d06bac 100644 --- a/packages/gateway-management-ui/src/gateways/gateway-data.ts +++ b/packages/gateway-management-ui/src/gateways/gateway-data.ts @@ -165,6 +165,27 @@ export function aggregateGatewayDisplayStatusCounts( return counts; } +/** Maps fleet-wide Prometheus phase counts into dashboard display buckets. */ +export function gatewayPhaseCountsToDisplayStatusCounts( + counts: Record, +): GatewayDisplayStatusCounts { + const displayCounts: GatewayDisplayStatusCounts = { + degraded: 0, + failed: 0, + healthy: 0, + provisioning: 0, + }; + + for (const phase of gatewayCanonicalPhaseStrings) { + const bucket = gatewayDisplayStatusBucket( + resolveGatewayDisplayStatus(phase, undefined), + ); + displayCounts[bucket] += counts[phase]; + } + + return displayCounts; +} + type GatewayConsoleRecord = Pick< GatewayRecord, "phase" | "status" | "externalDns" | "consoleUrl" diff --git a/packages/gateway-management-ui/src/index.ts b/packages/gateway-management-ui/src/index.ts index bae3f5f3e..a1871c9ab 100644 --- a/packages/gateway-management-ui/src/index.ts +++ b/packages/gateway-management-ui/src/index.ts @@ -61,6 +61,7 @@ export { aggregateGatewayDisplayStatusCounts, gatewayCanonicalPhaseStrings, gatewayCanonicalPhases, + gatewayPhaseCountsToDisplayStatusCounts, gatewayListQueryKey, gatewayListQueryRoot, gatewayPlacementBatchQueryKey, diff --git a/packages/gateway-management-ui/src/metrics/gateway-metrics-data.test.ts b/packages/gateway-management-ui/src/metrics/gateway-metrics-data.test.ts index 6fa36e3db..24670ec70 100644 --- a/packages/gateway-management-ui/src/metrics/gateway-metrics-data.test.ts +++ b/packages/gateway-management-ui/src/metrics/gateway-metrics-data.test.ts @@ -29,7 +29,7 @@ describe("fetchGatewayMetrics", () => { Degraded: 1, Failed: 0, }); - expect(fetch).toHaveBeenCalledWith("/api/hypershell/v1/metrics/gateways", { + expect(fetch).toHaveBeenCalledWith("/api/metrics/gateways", { credentials: "same-origin", signal: undefined, }); @@ -80,7 +80,7 @@ describe("fetchGatewayMetrics", () => { await fetchGatewayMetrics(controller.signal); - expect(fetch).toHaveBeenCalledWith("/api/hypershell/v1/metrics/gateways", { + expect(fetch).toHaveBeenCalledWith("/api/metrics/gateways", { credentials: "same-origin", signal: controller.signal, }); diff --git a/packages/gateway-management-ui/src/metrics/gateway-metrics-data.ts b/packages/gateway-management-ui/src/metrics/gateway-metrics-data.ts index a39befd6f..be247e2fc 100644 --- a/packages/gateway-management-ui/src/metrics/gateway-metrics-data.ts +++ b/packages/gateway-management-ui/src/metrics/gateway-metrics-data.ts @@ -22,7 +22,7 @@ export function emptyGatewayPhaseCounts(): GatewayPhaseCounts { export async function fetchGatewayMetrics( signal?: AbortSignal, ): Promise { - const response = await fetch("/api/hypershell/v1/metrics/gateways", { + const response = await fetch("/api/metrics/gateways", { credentials: "same-origin", signal, }); diff --git a/packages/operational-dashboard-ui/DATA_SOURCES.md b/packages/operational-dashboard-ui/DATA_SOURCES.md index 31e99de63..0623bf05a 100644 --- a/packages/operational-dashboard-ui/DATA_SOURCES.md +++ b/packages/operational-dashboard-ui/DATA_SOURCES.md @@ -10,23 +10,23 @@ Data is loaded through `useGetMetricsData` → `dashboard.getOperationalMetrics` ## Connected metrics -| Widget / metric ID | Source | Notes | -| --------------------------- | --------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `provisioned-gateways` | HyperShell API `GET /api/hypershell/v1/gateways` (paginated) | Display-status breakdown (`healthy`, `provisioning`, `degraded`, `failed`) using the same phase/status presentation rules as the gateway list. Total count. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | -| `gateway-status` | Same as `provisioned-gateways` | Uses the `status` field on the provisioned-gateways metric. | -| `provisioned-sandboxes` | HyperShell API `GET /api/hypershell/v1/gateways` (paginated) | Sum of `active_sandbox_count` across all gateways. Advisory control-plane field; omitted from the response when unset on a gateway. | -| `registered-users` | HyperShell API `GET /api/hypershell/v1/users` (`page=1`, `size=1`) | Total registered users from the List `total` field. Requires dashboard-operator authorization (`platform:admin` or `hypershell-admins`). Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | -| `memory` | BFF `GET /api/metrics/cluster-memory` (Prometheus) | Hub-cluster node memory from Prometheus node-exporter via `sum(node_memory_MemTotal_bytes)` (capacity) and `sum(node_memory_MemAvailable_bytes)` (available). Adapter maps used/capacity bytes to whole GiB for the utilization donut. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | -| `cpu` | BFF `GET /api/metrics/cluster-cpu` (Prometheus) | Hub-cluster node CPU from the same node-exporter DaemonSet as memory. Capacity: `sum(count by (instance) (node_cpu_seconds_total{mode="idle"}))`. Used: `sum(rate(node_cpu_seconds_total{mode!="idle"}[5m]))`. Adapter maps fractional used/capacity cores to whole cores for the utilization donut. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | -| `pods` | BFF `GET /api/metrics/cluster-pods` (Prometheus) | Hub-cluster pod capacity from kube-state-metrics. Capacity: `sum(kube_node_status_allocatable{resource="pods"})`. Used: `count(kube_pod_info)` (all phases while pod objects exist). Phase breakdown: `sum(kube_pod_status_phase{phase=""})`. Adapter maps phase fields to `podPhases`; `pods` widget uses `PodCapacityChart` with gray Unused segment (`total - value`). Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | -| `nodes` | BFF `GET /api/metrics/cluster-nodes` (Prometheus) | Hub-cluster node inventory from kube-state-metrics. Total: `count(kube_node_info)`. Ready: `sum(kube_node_status_condition{condition="Ready",status="true"})`. Adapter maps `ready_nodes` → `status.healthy` and `not_ready_nodes` → `status.failed`. `nodes` widget uses `NodeStatusChart`. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | -| `provision-time` | BFF `GET /api/metrics/gateway-provision-duration` (Prometheus) | Average in the system summary; average, P50, and P95 in the `provision-time` widget from `gateway_provision_duration_seconds` histogram per `platform/gateway-provision-time.spec.md`. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | -| `managed-clusters` | HyperShell API `GET /api/hypershell/v1/managed_clusters` (paginated) | Total registered managed clusters with `inventoryStatus`, `createdLast30Days`, `inventoryProviders`, and `inventoryRegions`. Page size 100, ordered by `name asc`. Requires dashboard-operator or `gateway:creator` authorization. A failed list request omits the `platform-inventory` source (OP-DASH-19). Refreshes every 15 minutes. | -| `managed-cluster-providers` | Same as `managed-clusters` | Provider donut from `inventoryProviders`; legend ordered by descending count. On the default layout (OP-DASH-20). | -| `managed-cluster-regions` | Same as `managed-clusters` | Placement donut from `inventoryRegions` (`{region} ({provider})` keys); legend ordered by descending count. On the default layout (OP-DASH-20). | -| `managed-databases` | HyperShell API `GET /api/hypershell/v1/managed_databases` (paginated) | Total registered managed databases with `inventoryStatus` breakdown. Same pagination and authorization rules as `managed-clusters`. Loaded in the same `platform-inventory` metric source. | -| `managed-database-status` | Same as `managed-databases` | Status donut from `inventoryStatus`; on the default layout (OP-DASH-20). | -| `inventory-summary` | Same as `managed-clusters` and `managed-databases` | `InventorySummaryCard` DescriptionList on the default layout (OP-DASH-20). Shows totals, 30-day cluster creations, and failure/warning status icons when present. | +| Widget / metric ID | Source | Notes | +| --------------------------- | -------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `provisioned-gateways` | BFF `GET /api/metrics/gateways` (Prometheus) | Fleet-wide gateway phase counts from `hypershell_gateways_total`. Display-status breakdown (`healthy`, `provisioning`, `degraded`, `failed`) via `gatewayPhaseCountsToDisplayStatusCounts`. Total count. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | +| `gateway-status` | Same as `provisioned-gateways` | Uses the `status` field on the provisioned-gateways metric. | +| `provisioned-sandboxes` | BFF `GET /api/metrics/gateway-sandboxes` (Prometheus) | Sum of `hypershell_gateways_active_sandboxes_total` across all gateways. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | +| `registered-users` | BFF `GET /api/metrics/registered-users` (Prometheus) | Total registered users from `hypershell_users_registered_total`. Requires `platform:admin`. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | +| `memory` | BFF `GET /api/metrics/cluster-memory` (Prometheus) | Hub-cluster node memory from Prometheus node-exporter via `sum(node_memory_MemTotal_bytes)` (capacity) and `sum(node_memory_MemAvailable_bytes)` (available). Adapter maps used/capacity bytes to whole GiB for the utilization donut. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | +| `cpu` | BFF `GET /api/metrics/cluster-cpu` (Prometheus) | Hub-cluster node CPU from the same node-exporter DaemonSet as memory. Capacity: `sum(count by (instance) (node_cpu_seconds_total{mode="idle"}))`. Used: `sum(rate(node_cpu_seconds_total{mode!="idle"}[5m]))`. Adapter maps fractional used/capacity cores to whole cores for the utilization donut. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | +| `pods` | BFF `GET /api/metrics/cluster-pods` (Prometheus) | Hub-cluster pod capacity from kube-state-metrics. Capacity: `sum(kube_node_status_allocatable{resource="pods"})`. Used: `count(kube_pod_info)` (all phases while pod objects exist). Phase breakdown: `sum(kube_pod_status_phase{phase=""})`. Adapter maps phase fields to `podPhases`; `pods` widget uses `PodCapacityChart` with gray Unused segment (`total - value`). Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | +| `nodes` | BFF `GET /api/metrics/cluster-nodes` (Prometheus) | Hub-cluster node inventory from kube-state-metrics. Total: `count(kube_node_info)`. Ready: `sum(kube_node_status_condition{condition="Ready",status="true"})`. Adapter maps `ready_nodes` → `status.healthy` and `not_ready_nodes` → `status.failed`. `nodes` widget uses `NodeStatusChart`. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | +| `provision-time` | BFF `GET /api/metrics/gateway-provision-duration` (Prometheus) | Average in the system summary; average, P50, and P95 in the `provision-time` widget from `gateway_provision_duration_seconds` histogram per `platform/gateway-provision-time.spec.md`. Refreshes every 15 minutes (`operationalDashboardRefreshMilliseconds`). | +| `managed-clusters` | BFF `GET /api/metrics/platform-inventory` (Prometheus) | Total registered managed clusters with `inventoryStatus`, `createdLast30Days`, `inventoryProviders`, and `inventoryRegions` from `hypershell_managed_clusters_*` metrics. Requires `platform:admin`. A failed query omits the `platform-inventory` source (OP-DASH-19). Refreshes every 15 minutes. | +| `managed-cluster-providers` | Same as `managed-clusters` | Provider donut from `inventoryProviders`; legend ordered by descending count. On the default layout (OP-DASH-20). | +| `managed-cluster-regions` | Same as `managed-clusters` | Placement donut from `inventoryRegions` (`{region} ({provider})` keys); legend ordered by descending count. On the default layout (OP-DASH-20). | +| `managed-databases` | Same as `managed-clusters` | Total registered managed databases with `inventoryStatus` breakdown from `hypershell_managed_databases_*` metrics. Loaded in the same `platform-inventory` metric source. | +| `managed-database-status` | Same as `managed-databases` | Status donut from `inventoryStatus`; on the default layout (OP-DASH-20). | +| `inventory-summary` | Same as `managed-clusters` and `managed-databases` | `InventorySummaryCard` DescriptionList on the default layout (OP-DASH-20). Shows totals, 30-day cluster creations, and failure/warning status icons when present. | ## Not connected (widgets remain, data unavailable) diff --git a/packages/operational-dashboard-ui/src/application/dashboard-types.ts b/packages/operational-dashboard-ui/src/application/dashboard-types.ts index 7e9e9cc3a..658e71336 100644 --- a/packages/operational-dashboard-ui/src/application/dashboard-types.ts +++ b/packages/operational-dashboard-ui/src/application/dashboard-types.ts @@ -53,7 +53,7 @@ export type DashboardMetricSourceId = | "cluster-memory" | "cluster-nodes" | "cluster-pods" - | "gateway-list" + | "gateway-metrics" | "platform-inventory" | "registered-users"; diff --git a/packages/operational-dashboard-ui/src/dashboard/dashboard-metric-sources.ts b/packages/operational-dashboard-ui/src/dashboard/dashboard-metric-sources.ts index 8838dc0c3..043243f2c 100644 --- a/packages/operational-dashboard-ui/src/dashboard/dashboard-metric-sources.ts +++ b/packages/operational-dashboard-ui/src/dashboard/dashboard-metric-sources.ts @@ -5,14 +5,14 @@ export type DashboardMetricSourceId = | "cluster-memory" | "cluster-nodes" | "cluster-pods" - | "gateway-list" + | "gateway-metrics" | "platform-inventory" | "registered-users"; export const DASHBOARD_METRIC_SOURCE_METRIC_IDS: Readonly< Record > = { - "gateway-list": [ + "gateway-metrics": [ "provisioned-gateways", "provisioned-sandboxes", "provision-time",