From efc64e82b9f6224ac7e390528c06db880c41fc2f Mon Sep 17 00:00:00 2001 From: Gianluca Mardente Date: Fri, 5 Jun 2026 13:48:32 +0200 Subject: [PATCH] Add per-cluster deployment health signals (hasIssues, isProvisioning) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The /capiclusters and /sveltosclusters endpoints now include two new boolean fields on each cluster: - hasIssues — true when at least one ClusterSummary has a feature in Failed or FailedNonRetriable state, or in Provisioning state with a non-empty failure message. The latter covers the Sveltos retry cycle where a retriable error moves status back to Provisioning without clearing the failure message; the message is only reset when the feature reaches Provisioned. - isProvisioning — true when at least one ClusterSummary is actively deploying (clean Provisioning with no failure message) or waiting for a dependency (status.dependencies non-empty). Always false when hasIssues is true, so the two signals are mutually exclusive from the consumer's perspective. Both signals are maintained as secondary in-memory indexes (clusterWithIssues and clusterProvisioning, each a map[string]*libsveltosset.Set) updated by the existing ClusterSummary watcher. No per-request scanning — each cluster lookup is O(1). Entries are created lazily and deleted when the set empties, keeping memory cost proportional to clusters with active state rather than total ClusterSummary count. --- go.mod | 8 +- go.sum | 20 +-- hack/tools/go.mod | 4 +- hack/tools/go.sum | 8 +- internal/server/http.go | 13 +- internal/server/managed_clusters.go | 8 +- internal/server/manager.go | 163 +++++++++++++++-- internal/server/manager_test.go | 259 ++++++++++++++++++++++++++++ 8 files changed, 440 insertions(+), 43 deletions(-) diff --git a/go.mod b/go.mod index e5bf884..6834b8e 100644 --- a/go.mod +++ b/go.mod @@ -4,12 +4,12 @@ go 1.26.3 require ( github.com/TwiN/go-color v1.4.1 - github.com/fluxcd/source-controller/api v1.8.4 + github.com/fluxcd/source-controller/api v1.8.5 github.com/gin-gonic/gin v1.12.0 github.com/go-logr/logr v1.4.3 - github.com/modelcontextprotocol/go-sdk v1.6.0 - github.com/onsi/ginkgo/v2 v2.28.3 - github.com/onsi/gomega v1.40.0 + github.com/modelcontextprotocol/go-sdk v1.6.1 + github.com/onsi/ginkgo/v2 v2.29.0 + github.com/onsi/gomega v1.41.0 github.com/pkg/errors v0.9.1 github.com/projectsveltos/addon-controller v1.10.0 github.com/projectsveltos/event-manager v1.10.0 diff --git a/go.sum b/go.sum index 606d33d..9a72bef 100644 --- a/go.sum +++ b/go.sum @@ -45,8 +45,8 @@ github.com/fluxcd/pkg/apis/acl v0.9.0 h1:wBpgsKT+jcyZEcM//OmZr9RiF8klL3ebrDp2u2T github.com/fluxcd/pkg/apis/acl v0.9.0/go.mod h1:TttNS+gocsGLwnvmgVi3/Yscwqrjc17+vhgYfqkfrV4= github.com/fluxcd/pkg/apis/meta v1.27.0 h1:EspByEk5j8w3rs1cGbEh9AjSmpDwQIz7DFG/zzqf6uI= github.com/fluxcd/pkg/apis/meta v1.27.0/go.mod h1:2t6JyrRfvIBhx6EBnXfFh/6sCCJ1db9WGaqko0JmNOE= -github.com/fluxcd/source-controller/api v1.8.4 h1:ZJIh7OFhjxZgsD81ahxxbu3ggA7qySIjKWbK5+PKmOI= -github.com/fluxcd/source-controller/api v1.8.4/go.mod h1:sio4t49RDx+S1etHRFAEEw8qfVuw0KKlOg8bRVlEYPM= +github.com/fluxcd/source-controller/api v1.8.5 h1:mLKc9YVMk46JCt1BQbkG6irkrpBZp95kiXh2+GYB6KQ= +github.com/fluxcd/source-controller/api v1.8.5/go.mod h1:sio4t49RDx+S1etHRFAEEw8qfVuw0KKlOg8bRVlEYPM= github.com/frankban/quicktest v1.14.6 h1:7Xjx+VpznH+oBnejlPUj8oUpdxnVs4f8XU8WnHkI4W8= github.com/frankban/quicktest v1.14.6/go.mod h1:4ptaffx2x8+WTWXmUCuVU6aPUX1/Mz7zb5vbUoiM6w0= github.com/fsnotify/fsnotify v1.9.0 h1:2Ml+OJNzbYCTzsxtv8vKSFD9PbJjmhYF14k/jKC7S9k= @@ -109,8 +109,6 @@ github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX github.com/google/gofuzz v1.0.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg= github.com/google/gofuzz v1.2.0 h1:xRy4A+RhZaiKjJ1bPfwQ8sedCA+YS2YcCHW6ec7JMi0= github.com/google/gofuzz v1.2.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg= -github.com/google/jsonschema-go v0.4.2 h1:tmrUohrwoLZZS/P3x7ex0WAVknEkBZM46iALbcqoRA8= -github.com/google/jsonschema-go v0.4.2/go.mod h1:r5quNTdLOYEz95Ru18zA0ydNbBuYoo9tgaYcxEYhJVE= github.com/google/jsonschema-go v0.4.3 h1:/DBOLZTfDow7pe2GmaJNhltueGTtDKICi8V8p+DQPd0= github.com/google/jsonschema-go v0.4.3/go.mod h1:r5quNTdLOYEz95Ru18zA0ydNbBuYoo9tgaYcxEYhJVE= github.com/google/pprof v0.0.0-20260402051712-545e8a4df936 h1:EwtI+Al+DeppwYX2oXJCETMO23COyaKGP6fHVpkpWpg= @@ -153,10 +151,8 @@ github.com/mitchellh/copystructure v1.2.0 h1:vpKXTN4ewci03Vljg/q9QvCGUDttBOGBIa1 github.com/mitchellh/copystructure v1.2.0/go.mod h1:qLl+cE2AmVv+CoeAwDPye/v+N2HKCj9FbZEVFJRxO9s= github.com/mitchellh/reflectwalk v1.0.2 h1:G2LzWKi524PWgd3mLHV8Y5k7s6XUvT0Gef6zxSIeXaQ= github.com/mitchellh/reflectwalk v1.0.2/go.mod h1:mSTlrgnPZtwu0c4WaC2kGObEpuNDbx0jmZXqmk4esnw= -github.com/modelcontextprotocol/go-sdk v1.5.0 h1:CHU0FIX9kpueNkxuYtfYQn1Z0slhFzBZuq+x6IiblIU= -github.com/modelcontextprotocol/go-sdk v1.5.0/go.mod h1:gggDIhoemhWs3BGkGwd1umzEXCEMMvAnhTrnbXJKKKA= -github.com/modelcontextprotocol/go-sdk v1.6.0 h1:PPLS3kn7WtOEnR+Af4X5H96SG0qSab8R/ZQT/HkhPkY= -github.com/modelcontextprotocol/go-sdk v1.6.0/go.mod h1:kzm3kzFL1/+AziGOE0nUs3gvPoNxMCvkxokMkuFapXQ= +github.com/modelcontextprotocol/go-sdk v1.6.1 h1:0zOSupjKUxPKSocPT1Wtago+mUHU2/uZ4xSOY0FGReU= +github.com/modelcontextprotocol/go-sdk v1.6.1/go.mod h1:kzm3kzFL1/+AziGOE0nUs3gvPoNxMCvkxokMkuFapXQ= github.com/modern-go/concurrent v0.0.0-20180228061459-e0a39a4cb421/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd h1:TRLaZ9cD/w8PVh93nsPXa1VrQ6jlwL5oN8l14QlcNfg= github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd/go.mod h1:6dJC0mAP4ikYIbvyc7fijjWJddQyLn8Ig3JB5CqoB9Q= @@ -165,10 +161,10 @@ github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee h1:W5t00kpgFd github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= -github.com/onsi/ginkgo/v2 v2.28.3 h1:4JvMdwtFU0imd8fHx25OJXoDMRexnf8v5NHKYSTTji4= -github.com/onsi/ginkgo/v2 v2.28.3/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= -github.com/onsi/gomega v1.40.0 h1:Vtol0e1MghCD2ZVIilPDIg44XSL9l2QAn8ZNaljWcJc= -github.com/onsi/gomega v1.40.0/go.mod h1:M/Uqpu/8qTjtzCLUA2zJHX9Iilrau25x1PdoSRbWh5A= +github.com/onsi/ginkgo/v2 v2.29.0 h1:rfh+ZFjgJhYWRoIqVf3Uwx/W20yLrcrE2h2GmYVRaag= +github.com/onsi/ginkgo/v2 v2.29.0/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= +github.com/onsi/gomega v1.41.0 h1:OwKp4pXNgVxf6sCplzYo794OFNuoL2q2SBMU5NSWOjA= +github.com/onsi/gomega v1.41.0/go.mod h1:M/Uqpu/8qTjtzCLUA2zJHX9Iilrau25x1PdoSRbWh5A= github.com/pelletier/go-toml/v2 v2.2.4 h1:mye9XuhQ6gvn5h28+VilKrrPoQVanw5PMw/TB0t5Ec4= github.com/pelletier/go-toml/v2 v2.2.4/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY= github.com/pkg/errors v0.9.1 h1:FEBLx1zS214owpjy7qsBeixbURkuhQAwrK5UwLGTwt4= diff --git a/hack/tools/go.mod b/hack/tools/go.mod index fb15487..3a841ab 100644 --- a/hack/tools/go.mod +++ b/hack/tools/go.mod @@ -4,12 +4,12 @@ go 1.26.3 require ( github.com/a8m/envsubst v1.4.3 - github.com/onsi/ginkgo/v2 v2.28.3 + github.com/onsi/ginkgo/v2 v2.29.0 golang.org/x/oauth2 v0.36.0 golang.org/x/tools v0.45.0 k8s.io/client-go v0.36.1 sigs.k8s.io/controller-tools v0.21.0 - sigs.k8s.io/kind v0.31.0 + sigs.k8s.io/kind v0.32.0 ) require ( diff --git a/hack/tools/go.sum b/hack/tools/go.sum index 55f0334..7b4302a 100644 --- a/hack/tools/go.sum +++ b/hack/tools/go.sum @@ -132,8 +132,8 @@ github.com/nxadm/tail v1.4.8 h1:nPr65rt6Y5JFSKQO7qToXr7pePgD6Gwiw05lkbyAQTE= github.com/nxadm/tail v1.4.8/go.mod h1:+ncqLTQzXmGhMZNUePPaPqPvBxHAIsmXswZKocGu+AU= github.com/onsi/ginkgo v1.16.5 h1:8xi0RTUf59SOSfEtZMvwTvXYMzG4gV23XVHOZiXNtnE= github.com/onsi/ginkgo v1.16.5/go.mod h1:+E8gABHa3K6zRBolWtd+ROzc/U5bkGt0FwiG042wbpU= -github.com/onsi/ginkgo/v2 v2.28.3 h1:4JvMdwtFU0imd8fHx25OJXoDMRexnf8v5NHKYSTTji4= -github.com/onsi/ginkgo/v2 v2.28.3/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= +github.com/onsi/ginkgo/v2 v2.29.0 h1:rfh+ZFjgJhYWRoIqVf3Uwx/W20yLrcrE2h2GmYVRaag= +github.com/onsi/ginkgo/v2 v2.29.0/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= github.com/onsi/gomega v1.40.0 h1:Vtol0e1MghCD2ZVIilPDIg44XSL9l2QAn8ZNaljWcJc= github.com/onsi/gomega v1.40.0/go.mod h1:M/Uqpu/8qTjtzCLUA2zJHX9Iilrau25x1PdoSRbWh5A= github.com/pelletier/go-toml v1.9.5 h1:4yBQzkHv+7BHq2PQUZF3Mx0IYxG7LsP222s7Agd3ve8= @@ -281,8 +281,8 @@ sigs.k8s.io/controller-tools v0.21.0 h1:KXDQza3bgjlPY6xLR63tI/40gzjhyUAvkCrwzd2/ sigs.k8s.io/controller-tools v0.21.0/go.mod h1:DLIypi3Q2+azVAP8jr/mHXJgveYYHFjhnNOUuBJ10JE= sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 h1:IpInykpT6ceI+QxKBbEflcR5EXP7sU1kvOlxwZh5txg= sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730/go.mod h1:mdzfpAEoE6DHQEN0uh9ZbOCuHbLK5wOm7dK4ctXE9Tg= -sigs.k8s.io/kind v0.31.0 h1:UcT4nzm+YM7YEbqiAKECk+b6dsvc/HRZZu9U0FolL1g= -sigs.k8s.io/kind v0.31.0/go.mod h1:FSqriGaoTPruiXWfRnUXNykF8r2t+fHtK0P0m1AbGF8= +sigs.k8s.io/kind v0.32.0 h1:p9hscbj98u/qyrjVpjId86LI70nQmbSsipV7wCG10Xk= +sigs.k8s.io/kind v0.32.0/go.mod h1:FSqriGaoTPruiXWfRnUXNykF8r2t+fHtK0P0m1AbGF8= sigs.k8s.io/randfill v1.0.0 h1:JfjMILfT8A6RbawdsK2JXGBR5AQVfd+9TbzrlneTyrU= sigs.k8s.io/randfill v1.0.0/go.mod h1:XeLlZ/jmk4i1HRopwe7/aU3H5n1zNUcX6TM94b3QxOY= sigs.k8s.io/structured-merge-diff/v6 v6.4.0 h1:qmp2e3ZfFi1/jJbDGpD4mt3wyp6PE1NfKHCYLqgNQJo= diff --git a/internal/server/http.go b/internal/server/http.go index 0acee87..bd849c7 100644 --- a/internal/server/http.go +++ b/internal/server/http.go @@ -119,7 +119,7 @@ var ( return } - managedClusterData := getManagedClusterData(clusters, filters) + managedClusterData := getManagedClusterData(clusters, filters, manager, libsveltosv1beta1.ClusterTypeCapi) sort.Sort(managedClusterData) result, err := getClustersInRange(managedClusterData, limit, skip) @@ -174,7 +174,7 @@ var ( return } - managedClusterData := getManagedClusterData(clusters, filters) + managedClusterData := getManagedClusterData(clusters, filters, manager, libsveltosv1beta1.ClusterTypeSveltos) sort.Sort(managedClusterData) result, err := getClustersInRange(managedClusterData, limit, skip) @@ -1054,6 +1054,7 @@ func (m *instance) start(ctx context.Context, port string, logger logr.Logger) { } func getManagedClusterData(clusters map[corev1.ObjectReference]ClusterInfo, filters *clusterFilters, + manager *instance, clusterType libsveltosv1beta1.ClusterType, ) ManagedClusters { data := make(ManagedClusters, 0) @@ -1077,9 +1078,11 @@ func getManagedClusterData(clusters map[corev1.ObjectReference]ClusterInfo, filt } data = append(data, ManagedCluster{ - Namespace: k.Namespace, - Name: k.Name, - ClusterInfo: clusters[k], + Namespace: k.Namespace, + Name: k.Name, + ClusterInfo: clusters[k], + HasIssues: manager.ClusterHasIssues(clusterType, k.Namespace, k.Name), + IsProvisioning: manager.ClusterIsProvisioning(clusterType, k.Namespace, k.Name), }) } diff --git a/internal/server/managed_clusters.go b/internal/server/managed_clusters.go index 8631f2e..640bbe3 100644 --- a/internal/server/managed_clusters.go +++ b/internal/server/managed_clusters.go @@ -25,9 +25,11 @@ import ( ) type ManagedCluster struct { - Namespace string `json:"namespace"` - Name string `json:"name"` - ClusterInfo `json:"clusterInfo"` + Namespace string `json:"namespace"` + Name string `json:"name"` + ClusterInfo `json:"clusterInfo"` + HasIssues bool `json:"hasIssues"` + IsProvisioning bool `json:"isProvisioning"` } type ManagedClusters []ManagedCluster diff --git a/internal/server/manager.go b/internal/server/manager.go index 7f4f334..db5ef4f 100644 --- a/internal/server/manager.go +++ b/internal/server/manager.go @@ -51,6 +51,9 @@ type ClusterProfileStatus struct { ClusterType libsveltosv1beta1.ClusterType `json:"clusterType"` ClusterName string `json:"clusterName"` Summary []ClusterFeatureSummary `json:"summary"` + // Dependencies is the human-readable status from ClusterSummary.Status.Dependencies. + // Non-empty means the ClusterSummary is waiting for one or more dependencies. + Dependencies string `json:"dependencies,omitempty"` } type ClusterFeatureSummary struct { @@ -93,7 +96,16 @@ type instance struct { sveltosClusters map[corev1.ObjectReference]ClusterInfo capiClusters map[corev1.ObjectReference]ClusterInfo clusterSummaryReport map[corev1.ObjectReference]ClusterProfileStatus - profiles map[corev1.ObjectReference]ProfileInfo + // key: "//" + // value: set of ClusterSummary ObjectReferences that have at least one failed feature + // Entry is created lazily and removed when the set becomes empty. + clusterWithIssues map[string]*libsveltosset.Set + // key: same format as clusterWithIssues + // value: set of ClusterSummary ObjectReferences that are actively provisioning + // (Provisioning with no failure message, or waiting for dependencies). + // Entry is created lazily and removed when the set becomes empty. + clusterProvisioning map[string]*libsveltosset.Set + profiles map[corev1.ObjectReference]ProfileInfo } var ( @@ -115,6 +127,8 @@ func InitializeManagerInstance(ctx context.Context, config *rest.Config, c clien sveltosClusters: make(map[corev1.ObjectReference]ClusterInfo), capiClusters: make(map[corev1.ObjectReference]ClusterInfo), clusterSummaryReport: make(map[corev1.ObjectReference]ClusterProfileStatus), + clusterWithIssues: make(map[string]*libsveltosset.Set), + clusterProvisioning: make(map[string]*libsveltosset.Set), profiles: make(map[corev1.ObjectReference]ProfileInfo), clusterMux: sync.RWMutex{}, clusterStatusesMux: sync.RWMutex{}, @@ -326,42 +340,165 @@ func (m *instance) AddClusterProfileStatus(summary *configv1beta1.ClusterSummary return } - // we're sure we're adding a proper cluster summary - // get the cluster profile name by using labels profileOwnerRef, err := configv1beta1.GetProfileOwnerReference(summary) if err != nil { return } - // initialize feature summaries slice clusterFeatureSummaries := MapToClusterFeatureSummaries(&summary.Status.FeatureSummaries) + dependencies := "" + if summary.Status.Dependencies != nil { + dependencies = *summary.Status.Dependencies + } + clusterProfileStatus := ClusterProfileStatus{ - ProfileName: profileOwnerRef.Name, - ProfileType: profileOwnerRef.Kind, - Namespace: summary.Namespace, - ClusterType: summary.Spec.ClusterType, - ClusterName: summary.Spec.ClusterName, - Summary: clusterFeatureSummaries, + ProfileName: profileOwnerRef.Name, + ProfileType: profileOwnerRef.Kind, + Namespace: summary.Namespace, + ClusterType: summary.Spec.ClusterType, + ClusterName: summary.Spec.ClusterName, + Summary: clusterFeatureSummaries, + Dependencies: dependencies, } + csRef := getKeyFromObject(m.scheme, summary) + clusterKey := clusterIssuesKey(summary.Spec.ClusterType, summary.Namespace, summary.Spec.ClusterName) + m.clusterStatusesMux.Lock() defer m.clusterStatusesMux.Unlock() - m.clusterSummaryReport[*getKeyFromObject(m.scheme, summary)] = clusterProfileStatus + m.clusterSummaryReport[*csRef] = clusterProfileStatus + m.updateClusterWithIssues(clusterKey, csRef, clusterSummaryHasIssues(clusterFeatureSummaries)) + m.updateClusterProvisioning(clusterKey, csRef, clusterSummaryIsProvisioning(clusterFeatureSummaries, dependencies)) } func (m *instance) RemoveClusterProfileStatus(summaryNamespace, summaryName string) { - clusterProfileStatus := &corev1.ObjectReference{ + csRef := &corev1.ObjectReference{ Namespace: summaryNamespace, Name: summaryName, Kind: configv1beta1.ClusterSummaryKind, APIVersion: configv1beta1.GroupVersion.String(), } + m.clusterStatusesMux.Lock() defer m.clusterStatusesMux.Unlock() - delete(m.clusterSummaryReport, *clusterProfileStatus) + if existing, ok := m.clusterSummaryReport[*csRef]; ok { + clusterKey := clusterIssuesKey(existing.ClusterType, existing.Namespace, existing.ClusterName) + m.updateClusterWithIssues(clusterKey, csRef, false) + m.updateClusterProvisioning(clusterKey, csRef, false) + } + + delete(m.clusterSummaryReport, *csRef) +} + +// ClusterHasIssues returns true if the cluster has at least one ClusterSummary with a failed feature. +func (m *instance) ClusterHasIssues(clusterType libsveltosv1beta1.ClusterType, clusterNamespace, clusterName string) bool { + m.clusterStatusesMux.RLock() + defer m.clusterStatusesMux.RUnlock() + + key := clusterIssuesKey(clusterType, clusterNamespace, clusterName) + set, ok := m.clusterWithIssues[key] + return ok && set.Len() > 0 +} + +// clusterIssuesKey returns the map key for clusterWithIssues. +func clusterIssuesKey(clusterType libsveltosv1beta1.ClusterType, clusterNamespace, clusterName string) string { + return fmt.Sprintf("%s/%s/%s", clusterType, clusterNamespace, clusterName) +} + +// clusterSummaryHasIssues returns true if any feature shows an issue. +// Two cases are covered: +// 1. Status is explicitly Failed or FailedNonRetriable. +// 2. Status is Provisioning but FailureMessage is non-empty — Sveltos retries retriable +// errors by moving status back to Provisioning without clearing the failure message; +// the message is only cleared when the feature reaches Provisioned. +func clusterSummaryHasIssues(features []ClusterFeatureSummary) bool { + for i := range features { + f := features[i] + if f.Status == libsveltosv1beta1.FeatureStatusFailed || + f.Status == libsveltosv1beta1.FeatureStatusFailedNonRetriable { + + return true + } + if f.Status == libsveltosv1beta1.FeatureStatusProvisioning && + f.FailureMessage != nil && *f.FailureMessage != "" { + + return true + } + } + return false +} + +// updateClusterWithIssues inserts or erases csRef from the per-cluster issues set. +// Must be called with clusterStatusesMux held for writing. +func (m *instance) updateClusterWithIssues(clusterKey string, csRef *corev1.ObjectReference, hasFailed bool) { + if hasFailed { + if m.clusterWithIssues[clusterKey] == nil { + m.clusterWithIssues[clusterKey] = &libsveltosset.Set{} + } + m.clusterWithIssues[clusterKey].Insert(csRef) + } else { + if set, ok := m.clusterWithIssues[clusterKey]; ok { + set.Erase(csRef) + if set.Len() == 0 { + delete(m.clusterWithIssues, clusterKey) + } + } + } +} + +// ClusterIsProvisioning returns true when the cluster has at least one ClusterSummary that +// is actively deploying (Provisioning with no failure) or waiting for dependencies, AND +// has no issues. Issues take precedence — a cluster with failures is not also "provisioning". +func (m *instance) ClusterIsProvisioning(clusterType libsveltosv1beta1.ClusterType, clusterNamespace, clusterName string) bool { + m.clusterStatusesMux.RLock() + defer m.clusterStatusesMux.RUnlock() + + key := clusterIssuesKey(clusterType, clusterNamespace, clusterName) + if set, ok := m.clusterWithIssues[key]; ok && set.Len() > 0 { + return false + } + set, ok := m.clusterProvisioning[key] + return ok && set.Len() > 0 +} + +// clusterSummaryIsProvisioning returns true when the ClusterSummary is actively provisioning +// without errors. Two conditions qualify: +// 1. Any feature has Provisioning status with no failure message (clean in-progress deploy). +// 2. The dependencies field is non-empty (waiting for another ClusterProfile to finish). +func clusterSummaryIsProvisioning(features []ClusterFeatureSummary, dependencies string) bool { + if dependencies != "" { + return true + } + for i := range features { + f := features[i] + if f.Status == libsveltosv1beta1.FeatureStatusProvisioning && + (f.FailureMessage == nil || *f.FailureMessage == "") { + + return true + } + } + return false +} + +// updateClusterProvisioning inserts or erases csRef from the per-cluster provisioning set. +// Must be called with clusterStatusesMux held for writing. +func (m *instance) updateClusterProvisioning(clusterKey string, csRef *corev1.ObjectReference, isProvisioning bool) { + if isProvisioning { + if m.clusterProvisioning[clusterKey] == nil { + m.clusterProvisioning[clusterKey] = &libsveltosset.Set{} + } + m.clusterProvisioning[clusterKey].Insert(csRef) + } else { + if set, ok := m.clusterProvisioning[clusterKey]; ok { + set.Erase(csRef) + if set.Len() == 0 { + delete(m.clusterProvisioning, clusterKey) + } + } + } } // getKeyFromObject returns the Key that can be used in the internal reconciler maps. diff --git a/internal/server/manager_test.go b/internal/server/manager_test.go index e46435b..5633913 100644 --- a/internal/server/manager_test.go +++ b/internal/server/manager_test.go @@ -410,6 +410,265 @@ var _ = Describe("Manager", func() { Expect(clusterProfileStatuses[0].ProfileName == properClusterSummary.Name).To(BeTrue()) }) + It("ClusterHasIssues returns true when Provisioning with a non-empty failure message (retry cycle)", func() { + // Sveltos retries retriable failures by moving status back to Provisioning + // without clearing the failure message. The message is only cleared on Provisioned. + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + failMsg := "health check failed: replicas not ready" + cs := createTestClusterSummary("cs-retry-cycle", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusProvisioning, FailureMessage: &failMsg}, + }, + ) + manager.AddClusterProfileStatus(cs) + + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + }) + + It("ClusterHasIssues returns false when Provisioning with no failure message", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + cs := createTestClusterSummary("cs-clean-provisioning", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusProvisioning}, + }, + ) + manager.AddClusterProfileStatus(cs) + + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + + It("ClusterHasIssues returns true when a ClusterSummary has a Failed feature", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + failMsg := "deploy error" + cs := createTestClusterSummary("cs-failed", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusFailed, FailureMessage: &failMsg}, + }, + ) + manager.AddClusterProfileStatus(cs) + + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + }) + + It("ClusterHasIssues returns true when a ClusterSummary has a FailedNonRetriable feature", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + cs := createTestClusterSummary("cs-failed-nr", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Resources", Status: libsveltosv1beta1.FeatureStatusFailedNonRetriable}, + }, + ) + manager.AddClusterProfileStatus(cs) + + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + }) + + It("ClusterHasIssues clears when a failed ClusterSummary recovers to Provisioned", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + failMsg := "transient error" + cs := createTestClusterSummary("cs-recover", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusFailed, FailureMessage: &failMsg}, + }, + ) + manager.AddClusterProfileStatus(cs) + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + + // Same ClusterSummary, now provisioned — issues must clear + cs.Status.FeatureSummaries[0].Status = libsveltosv1beta1.FeatureStatusProvisioned + manager.AddClusterProfileStatus(cs) + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + + It("ClusterHasIssues clears when the failed ClusterSummary is removed", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + failMsg := "deploy failed" + cs := createTestClusterSummary("cs-remove", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusFailed, FailureMessage: &failMsg}, + }, + ) + manager.AddClusterProfileStatus(cs) + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + + manager.RemoveClusterProfileStatus(cs.Namespace, cs.Name) + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + + It("ClusterHasIssues remains true when only one of two failed ClusterSummaries recovers", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + failMsg := "profile deploy failed" + cs1 := createTestClusterSummary("cs-multi-1", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusFailed, FailureMessage: &failMsg}, + }, + ) + cs2 := createTestClusterSummary("cs-multi-2", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Resources", Status: libsveltosv1beta1.FeatureStatusFailed, FailureMessage: &failMsg}, + }, + ) + manager.AddClusterProfileStatus(cs1) + manager.AddClusterProfileStatus(cs2) + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + + // cs1 recovers — cs2 still failing, flag must stay true + cs1.Status.FeatureSummaries[0].Status = libsveltosv1beta1.FeatureStatusProvisioned + manager.AddClusterProfileStatus(cs1) + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + + // cs2 also recovers — now clear + cs2.Status.FeatureSummaries[0].Status = libsveltosv1beta1.FeatureStatusProvisioned + manager.AddClusterProfileStatus(cs2) + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + + It("ClusterIsProvisioning returns true when a ClusterSummary is cleanly Provisioning", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + cs := createTestClusterSummary("cs-provisioning-clean", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusProvisioning}, + }, + ) + manager.AddClusterProfileStatus(cs) + + Expect(manager.ClusterIsProvisioning(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + + It("ClusterIsProvisioning returns true when a ClusterSummary is waiting for dependencies", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + depMsg := "ClusterProfile deploy-cert-manager is a dependency and it is not fully deployed yet" + cs := createTestClusterSummary("cs-waiting-deps", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{}, + ) + cs.Status.Dependencies = &depMsg + manager.AddClusterProfileStatus(cs) + + Expect(manager.ClusterIsProvisioning(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + }) + + It("ClusterIsProvisioning returns false when issues are present (issues take priority)", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + failMsg := "helm deploy error" + cs := createTestClusterSummary("cs-failed-not-provisioning", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusFailed, FailureMessage: &failMsg}, + }, + ) + manager.AddClusterProfileStatus(cs) + + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + Expect(manager.ClusterIsProvisioning(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + + It("ClusterIsProvisioning returns false and ClusterHasIssues returns true when Provisioning with failure message", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + failMsg := "health check: replicas not matching" + cs := createTestClusterSummary("cs-retry-not-provisioning", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusProvisioning, FailureMessage: &failMsg}, + }, + ) + manager.AddClusterProfileStatus(cs) + + Expect(manager.ClusterHasIssues(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + Expect(manager.ClusterIsProvisioning(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + + It("ClusterIsProvisioning clears when ClusterSummary reaches Provisioned", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + cs := createTestClusterSummary("cs-prov-to-done", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusProvisioning}, + }, + ) + manager.AddClusterProfileStatus(cs) + Expect(manager.ClusterIsProvisioning(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + + cs.Status.FeatureSummaries[0].Status = libsveltosv1beta1.FeatureStatusProvisioned + manager.AddClusterProfileStatus(cs) + Expect(manager.ClusterIsProvisioning(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + + It("ClusterIsProvisioning clears when ClusterSummary is removed", func() { + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() + + server.InitializeManagerInstance(ctx, nil, c, scheme, randomPort(), logger) + manager := server.GetManagerInstance() + + cs := createTestClusterSummary("cs-prov-removed", cluster.Namespace, cluster.Namespace, cluster.Name, + []configv1beta1.FeatureSummary{ + {FeatureID: "Helm", Status: libsveltosv1beta1.FeatureStatusProvisioning}, + }, + ) + manager.AddClusterProfileStatus(cs) + Expect(manager.ClusterIsProvisioning(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeTrue()) + + manager.RemoveClusterProfileStatus(cs.Namespace, cs.Name) + Expect(manager.ClusterIsProvisioning(libsveltosv1beta1.ClusterTypeCapi, cluster.Namespace, cluster.Name)).To(BeFalse()) + }) + It("AddProfile adds a profile and update dependencies", func() { ctx, cancel := context.WithCancel(context.Background()) defer cancel()