From 1d38665ddcf21526eb6c407339c39679a3b34121 Mon Sep 17 00:00:00 2001 From: Lee Calcote Date: Sat, 4 Jul 2026 19:33:30 -0500 Subject: [PATCH 1/2] docs: correct stale output-mode and informer-filtering claims; drop stray manifests - README named the default output mode "nats"; the -output flag value is "broker" (internal/config/types.go). Rename and clarify that broker mode publishes to Meshery Broker over NATS. - docs/agent-instructions/architecture.md claimed the dynamic informer factory filters via a label selector derived from the informer blacklist; GetListOptionsFunc is a deliberate no-op and filtering happens in the watch-list config (internal/config/crd_config.go). - Remove cert-manager pod dumps accidentally committed at the repo root (all-certmgs.yaml, cert-manager-cainjector-*.yaml, cert-manager-webhook-*.yaml); referenced nowhere. Signed-off-by: Lee Calcote --- README.md | 2 +- all-certmgs.yaml | 0 cert-manager-cainjector-c7d4dbdd9-jlstt.yaml | 139 --------------- cert-manager-webhook-847d7676c9-89rtq.yaml | 171 ------------------- docs/agent-instructions/architecture.md | 2 +- 5 files changed, 2 insertions(+), 312 deletions(-) delete mode 100644 all-certmgs.yaml delete mode 100644 cert-manager-cainjector-c7d4dbdd9-jlstt.yaml delete mode 100644 cert-manager-webhook-847d7676c9-89rtq.yaml diff --git a/README.md b/README.md index bb92e530..1c216cb9 100644 --- a/README.md +++ b/README.md @@ -37,7 +37,7 @@ alt="Meshery Logo" width="70%" />

MeshSync is Meshery's event-driven, continuous discovery and synchronization engine. It ensures that the configuration and operational state of Kubernetes (and any supported Meshery platform) are known to Meshery Server. When deployed into a Kubernetes cluster, MeshSync runs as a custom controller under the control of [Meshery Operator](https://docs.meshery.io/concepts/architecture/operator) and publishes resource changes over Meshery Broker (NATS). -MeshSync runs in one of two modes: **nats** (default - publishes Kubernetes resource events to NATS) and **file** (writes deduplicated cluster snapshots to disk, with no NATS or CRD dependency). Run `meshsync --help` for input parameters. +MeshSync runs in one of two output modes: **broker** (default - publishes Kubernetes resource events to Meshery Broker over NATS) and **file** (writes deduplicated cluster snapshots to disk, with no NATS or CRD dependency). Run `meshsync --help` for input parameters. ## Documentation diff --git a/all-certmgs.yaml b/all-certmgs.yaml deleted file mode 100644 index e69de29b..00000000 diff --git a/cert-manager-cainjector-c7d4dbdd9-jlstt.yaml b/cert-manager-cainjector-c7d4dbdd9-jlstt.yaml deleted file mode 100644 index 723375ba..00000000 --- a/cert-manager-cainjector-c7d4dbdd9-jlstt.yaml +++ /dev/null @@ -1,139 +0,0 @@ -apiVersion: v1 -kind: Pod -metadata: - creationTimestamp: "2024-04-01T04:29:32Z" - generateName: cert-manager-cainjector-c7d4dbdd9- - labels: - app: cainjector - app.kubernetes.io/component: cainjector - app.kubernetes.io/instance: cert-manager - app.kubernetes.io/managed-by: Helm - app.kubernetes.io/name: cainjector - app.kubernetes.io/version: v1.14.4 - helm.sh/chart: cert-manager-v1.14.4 - pod-template-hash: c7d4dbdd9 - name: cert-manager-cainjector-c7d4dbdd9-jlstt - namespace: cert-manager - ownerReferences: - - apiVersion: apps/v1 - blockOwnerDeletion: true - controller: true - kind: ReplicaSet - name: cert-manager-cainjector-c7d4dbdd9 - uid: 4ad5f4b4-6ddc-4bad-8208-418435b3b8d7 - resourceVersion: "4476" - uid: b905aeb8-3633-4ca9-b350-df3a87cdd60e -spec: - containers: - - args: - - --v=2 - - --leader-election-namespace=kube-system - env: - - name: POD_NAMESPACE - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: metadata.namespace - image: quay.io/jetstack/cert-manager-cainjector:v1.14.4 - imagePullPolicy: IfNotPresent - name: cert-manager-cainjector - resources: {} - securityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - readOnlyRootFilesystem: true - terminationMessagePath: /dev/termination-log - terminationMessagePolicy: File - volumeMounts: - - mountPath: /var/run/secrets/kubernetes.io/serviceaccount - name: kube-api-access-sktmn - readOnly: true - dnsPolicy: ClusterFirst - enableServiceLinks: false - nodeName: c3-medium-x86-03-meshery - nodeSelector: - kubernetes.io/os: linux - preemptionPolicy: PreemptLowerPriority - priority: 0 - restartPolicy: Always - schedulerName: default-scheduler - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - serviceAccount: cert-manager-cainjector - serviceAccountName: cert-manager-cainjector - terminationGracePeriodSeconds: 30 - tolerations: - - effect: NoExecute - key: node.kubernetes.io/not-ready - operator: Exists - tolerationSeconds: 300 - - effect: NoExecute - key: node.kubernetes.io/unreachable - operator: Exists - tolerationSeconds: 300 - volumes: - - name: kube-api-access-sktmn - projected: - defaultMode: 420 - sources: - - serviceAccountToken: - expirationSeconds: 3607 - path: token - - configMap: - items: - - key: ca.crt - path: ca.crt - name: kube-root-ca.crt - - downwardAPI: - items: - - fieldRef: - apiVersion: v1 - fieldPath: metadata.namespace - path: namespace -status: - conditions: - - lastProbeTime: null - lastTransitionTime: "2024-04-01T04:29:34Z" - status: "True" - type: PodReadyToStartContainers - - lastProbeTime: null - lastTransitionTime: "2024-04-01T04:29:32Z" - status: "True" - type: Initialized - - lastProbeTime: null - lastTransitionTime: "2024-04-01T04:29:34Z" - status: "True" - type: Ready - - lastProbeTime: null - lastTransitionTime: "2024-04-01T04:29:34Z" - status: "True" - type: ContainersReady - - lastProbeTime: null - lastTransitionTime: "2024-04-01T04:29:32Z" - status: "True" - type: PodScheduled - containerStatuses: - - containerID: containerd://1c7b0a11d4c38a0fce9c3abf2b69358eae486bf383344a18b730db6bf5e135a4 - image: quay.io/jetstack/cert-manager-cainjector:v1.14.4 - imageID: quay.io/jetstack/cert-manager-cainjector@sha256:30286297e5b4b71a86759d297a8109c6a1649fdc68d28f618d87edf12a2da417 - lastState: {} - name: cert-manager-cainjector - ready: true - restartCount: 0 - started: true - state: - running: - startedAt: "2024-04-01T04:29:33Z" - hostIP: 139.178.83.85 - hostIPs: - - ip: 139.178.83.85 - phase: Running - podIP: 192.168.0.14 - podIPs: - - ip: 192.168.0.14 - qosClass: BestEffort - startTime: "2024-04-01T04:29:32Z" diff --git a/cert-manager-webhook-847d7676c9-89rtq.yaml b/cert-manager-webhook-847d7676c9-89rtq.yaml deleted file mode 100644 index 57baf1e0..00000000 --- a/cert-manager-webhook-847d7676c9-89rtq.yaml +++ /dev/null @@ -1,171 +0,0 @@ -apiVersion: v1 -kind: Pod -metadata: - creationTimestamp: "2024-04-01T04:29:32Z" - generateName: cert-manager-webhook-847d7676c9- - labels: - app: webhook - app.kubernetes.io/component: webhook - app.kubernetes.io/instance: cert-manager - app.kubernetes.io/managed-by: Helm - app.kubernetes.io/name: webhook - app.kubernetes.io/version: v1.14.4 - helm.sh/chart: cert-manager-v1.14.4 - pod-template-hash: 847d7676c9 - name: cert-manager-webhook-847d7676c9-89rtq - namespace: cert-manager - ownerReferences: - - apiVersion: apps/v1 - blockOwnerDeletion: true - controller: true - kind: ReplicaSet - name: cert-manager-webhook-847d7676c9 - uid: 9ca18977-52e3-474e-93e7-cbcaba9a13fa - resourceVersion: "47870864" - uid: 4b127dff-feeb-4962-81b2-27b87d2a1418 -spec: - containers: - - args: - - --v=2 - - --secure-port=10250 - - --dynamic-serving-ca-secret-namespace=$(POD_NAMESPACE) - - --dynamic-serving-ca-secret-name=cert-manager-webhook-ca - - --dynamic-serving-dns-names=cert-manager-webhook - - --dynamic-serving-dns-names=cert-manager-webhook.$(POD_NAMESPACE) - - --dynamic-serving-dns-names=cert-manager-webhook.$(POD_NAMESPACE).svc - env: - - name: POD_NAMESPACE - valueFrom: - fieldRef: - apiVersion: v1 - fieldPath: metadata.namespace - image: quay.io/jetstack/cert-manager-webhook:v1.14.4 - imagePullPolicy: IfNotPresent - livenessProbe: - failureThreshold: 3 - httpGet: - path: /livez - port: 6080 - scheme: HTTP - initialDelaySeconds: 60 - periodSeconds: 10 - successThreshold: 1 - timeoutSeconds: 1 - name: cert-manager-webhook - ports: - - containerPort: 10250 - name: https - protocol: TCP - - containerPort: 6080 - name: healthcheck - protocol: TCP - readinessProbe: - failureThreshold: 3 - httpGet: - path: /healthz - port: 6080 - scheme: HTTP - initialDelaySeconds: 5 - periodSeconds: 5 - successThreshold: 1 - timeoutSeconds: 1 - resources: {} - securityContext: - allowPrivilegeEscalation: false - capabilities: - drop: - - ALL - readOnlyRootFilesystem: true - terminationMessagePath: /dev/termination-log - terminationMessagePolicy: File - volumeMounts: - - mountPath: /var/run/secrets/kubernetes.io/serviceaccount - name: kube-api-access-9657m - readOnly: true - dnsPolicy: ClusterFirst - enableServiceLinks: false - nodeName: c3-medium-x86-03-meshery - nodeSelector: - kubernetes.io/os: linux - preemptionPolicy: PreemptLowerPriority - priority: 0 - restartPolicy: Always - schedulerName: default-scheduler - securityContext: - runAsNonRoot: true - seccompProfile: - type: RuntimeDefault - serviceAccount: cert-manager-webhook - serviceAccountName: cert-manager-webhook - terminationGracePeriodSeconds: 30 - tolerations: - - effect: NoExecute - key: node.kubernetes.io/not-ready - operator: Exists - tolerationSeconds: 300 - - effect: NoExecute - key: node.kubernetes.io/unreachable - operator: Exists - tolerationSeconds: 300 - volumes: - - name: kube-api-access-9657m - projected: - defaultMode: 420 - sources: - - serviceAccountToken: - expirationSeconds: 3607 - path: token - - configMap: - items: - - key: ca.crt - path: ca.crt - name: kube-root-ca.crt - - downwardAPI: - items: - - fieldRef: - apiVersion: v1 - fieldPath: metadata.namespace - path: namespace -status: - conditions: - - lastProbeTime: null - lastTransitionTime: "2024-04-01T04:29:34Z" - status: "True" - type: PodReadyToStartContainers - - lastProbeTime: null - lastTransitionTime: "2024-04-01T04:29:32Z" - status: "True" - type: Initialized - - lastProbeTime: null - lastTransitionTime: "2024-11-06T05:50:58Z" - status: "True" - type: Ready - - lastProbeTime: null - lastTransitionTime: "2024-11-06T05:50:58Z" - status: "True" - type: ContainersReady - - lastProbeTime: null - lastTransitionTime: "2024-04-01T04:29:32Z" - status: "True" - type: PodScheduled - containerStatuses: - - containerID: containerd://f60e25f3ca9dddc430ade42179a757868cf2264e52682a003fb2fd52df7b8efc - image: quay.io/jetstack/cert-manager-webhook:v1.14.4 - imageID: quay.io/jetstack/cert-manager-webhook@sha256:11f7e7c462da3c0329e0a1e695a7bd37d6b3c28312d4edd4cc8d36f70ecbfa63 - lastState: {} - name: cert-manager-webhook - ready: true - restartCount: 0 - started: true - state: - running: - startedAt: "2024-04-01T04:29:33Z" - hostIP: 139.178.83.85 - hostIPs: - - ip: 139.178.83.85 - phase: Running - podIP: 192.168.0.15 - podIPs: - - ip: 192.168.0.15 - qosClass: BestEffort - startTime: "2024-04-01T04:29:32Z" diff --git a/docs/agent-instructions/architecture.md b/docs/agent-instructions/architecture.md index 6d363cd8..891111c0 100644 --- a/docs/agent-instructions/architecture.md +++ b/docs/agent-instructions/architecture.md @@ -29,7 +29,7 @@ main.go --parses CLI flags--> pkg/lib/meshsync.Run(...) - `main.go` parses flags (`-output`, `-outputFile`, `-outputNamespaces`, `-outputResources`, `-stopAfter`) and calls `pkg/lib/meshsync.Run(...)`. - `meshsync.Handler` (`meshsync/meshsync.go`) holds the config, logger, broker handle, dynamic informer factory, kube client, channel pool, output writer, and output-filtration config. `meshsync.New(...)` wires them together and derives the cluster ID via `pkg/utils.GetClusterID`. -- `GetDynamicInformer` builds a `dynamicinformer.DynamicSharedInformerFactory` filtered by a label selector derived from the config's informer blacklist. +- `GetDynamicInformer` builds a `dynamicinformer.DynamicSharedInformerFactory`. Resource filtering happens in the watch-list config (`internal/config/crd_config.go` decides which informers get registered); the factory's list-options hook (`GetListOptionsFunc`) is a deliberate no-op. ## Discovery Pipeline (`internal/pipeline`) From ed838612e6d89072513b4a6fa717ed7612cf5de7 Mon Sep 17 00:00:00 2001 From: Lee Calcote Date: Sat, 4 Jul 2026 19:52:16 -0500 Subject: [PATCH 2/2] docs: sweep remaining stale output-mode and list-options references Review follow-up: align the -outputNamespaces/-outputResources flag help text and the determineUseCRDFlag comment with the actual broker output mode (and drop the reference to a nonexistent channel output mode), and correct fd5-periodic-reconciliation.md where it still described GetListOptionsFunc as a blacklist label selector. Signed-off-by: Lee Calcote --- docs/design/fd5-periodic-reconciliation.md | 4 ++-- main.go | 4 ++-- pkg/lib/meshsync/meshsync.go | 4 ++-- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/docs/design/fd5-periodic-reconciliation.md b/docs/design/fd5-periodic-reconciliation.md index 9ecd4ee0..f8bb175f 100644 --- a/docs/design/fd5-periodic-reconciliation.md +++ b/docs/design/fd5-periodic-reconciliation.md @@ -107,7 +107,7 @@ Implemented as the third branch above: for each reconciled GVR, take the key set - **`meshsync/meshsync.go`**: `Handler` gets no new fields — a reconcile loop is a *behavior* of `Run()`, not new handler state, keeping with `pipeline.New`'s per-call-fresh philosophy the architecture doc already mandates. - **`meshsync/handlers.go`** — the core of the feature: - **New unexported function `(h *Handler) reconcileLoop(intervalStr string, stopCh channels.StopChannel)`**, started as a goroutine from `Run()` (alongside the existing `startAndTrackDiscovery()` call, `handlers.go:57-58`) only when the resolved interval is non-zero. On each tick (a `time.Ticker`, jittered per §"scale" below), it iterates `h.stores` (the same map `startDiscovery` populates at `discovery.go:48` — `h.stores[gvr-name] = cache.Store`) and, for each GVR present, performs: - 1. A fresh, filtered LIST against `h.kubeClient.DynamicKubeClient` for that GVR (respecting the same `GetListOptionsFunc` blacklist label selector already used for informer construction, `meshsync/meshsync.go:34-53`, so reconcile never re-lists resource kinds the informer itself was configured to exclude). + 1. A fresh, filtered LIST against `h.kubeClient.DynamicKubeClient` for that GVR (reconcile iterates only the GVRs whose informer pipelines were registered from the watch-list config; `GetListOptionsFunc` is a deliberate no-op, so there is no list-time label filtering to rely on). 2. Diff against `store.List()`/`store.ListKeys()` per §3(b)/(c). 3. Publish only drifted objects through the *existing* `internal/pipeline.RegisterInformer.publishItem`-equivalent path — since `publishItem` is a method on `RegisterInformer` (pipeline-internal, not exported), the cleanest reuse is to export a small helper in `internal/pipeline` (e.g. `pipeline.PublishDrift(log, outputWriter, clusterID, outputFiltration, cfg PipelineConfig, obj *unstructured.Unstructured, evtype broker.EventType) error`) that wraps the existing `model.ParseList` + `checkMustSkip` + `outputWriter.Write` sequence (lines 93-111 of `internal/pipeline/handlers.go`), called from both `RegisterInformer.publishItem` (refactored to delegate to it) and the new reconcile loop — avoiding duplicating the skip/filter/write logic in two places, which the CLAUDE.md "consistency with existing patterns... migrate the existing code accordingly" directive requires rather than a second parallel implementation. - Guard the whole loop on the `channels.Stop` channel exactly like `WatchCRDs` does (`handlers.go:359-361`) for clean shutdown. @@ -151,7 +151,7 @@ The default-off interval (§3(d)) **is** the feature flag — no separate boolea ## 8. Risks, Failure Modes, Perf/Scale -- **Re-list storm on large clusters.** A naive "list every configured GVR every tick" is the primary named risk. Mitigations designed in: (1) default-off (§3(d)); (2) enforced floor interval (5m minimum, §3(d)); (3) intra-tick staggering across GVRs with bounded concurrency (§4); (4) fleet-wide jitter via the existing `jitter()` helper (§4) so many clusters' MeshSync instances don't synchronize; (5) reconcile scope is **exactly** the informer's existing configured GVR set filtered through `GetListOptionsFunc`'s blacklist label selector (§4) — it never lists more than what's already being watched, and it respects `-outputResources` (via `outputFiltration.ResourceSet`, already applied at the pipeline-construction/write layer) so a restricted deployment reconciles only its restricted resource set. **`-outputNamespaces` is a write-time filter, not an informer/LIST-time scope** (§2) — a genuinely namespace-scoped reconcile (listing only within the allowed namespaces at the API-server level, cheaper than a cluster-wide LIST filtered client-side) is not achievable without deeper informer-factory changes (`NewFilteredDynamicSharedInformerFactory` takes one cluster-wide-or-single-namespace argument, not a set) — flagged explicitly as an out-of-scope enhancement in §11, with the interim mitigation being that write-time filtering still means a namespace-restricted deployment publishes no more drift than it would publish live events, even though its LIST cost is unreduced. +- **Re-list storm on large clusters.** A naive "list every configured GVR every tick" is the primary named risk. Mitigations designed in: (1) default-off (§3(d)); (2) enforced floor interval (5m minimum, §3(d)); (3) intra-tick staggering across GVRs with bounded concurrency (§4); (4) fleet-wide jitter via the existing `jitter()` helper (§4) so many clusters' MeshSync instances don't synchronize; (5) reconcile scope is **exactly** the informer's existing configured GVR set as registered from the watch-list config (§4; `GetListOptionsFunc` is a deliberate no-op) — it never lists more than what's already being watched, and it respects `-outputResources` (via `outputFiltration.ResourceSet`, already applied at the pipeline-construction/write layer) so a restricted deployment reconciles only its restricted resource set. **`-outputNamespaces` is a write-time filter, not an informer/LIST-time scope** (§2) — a genuinely namespace-scoped reconcile (listing only within the allowed namespaces at the API-server level, cheaper than a cluster-wide LIST filtered client-side) is not achievable without deeper informer-factory changes (`NewFilteredDynamicSharedInformerFactory` takes one cluster-wide-or-single-namespace argument, not a set) — flagged explicitly as an out-of-scope enhancement in §11, with the interim mitigation being that write-time filtering still means a namespace-restricted deployment publishes no more drift than it would publish live events, even though its LIST cost is unreduced. - **Dedup/RV-suppression interaction failing silently.** If the reconcile loop's own RV-diff (§3(b)) has a bug and always treats "unchanged" as "changed," it defeats the entire purpose and re-floods Server every tick. Test plan (§9) includes an explicit "steady state produces zero publishes" assertion, not just a "drift produces N publishes" assertion — the null case is equally load-bearing. - **Missed-delete false positives from partial/paginated or rate-limited LIST failures.** A LIST that fails midway or is truncated must never be diffed as "everything not returned is deleted" — a failed/partial LIST must abort that GVR's reconcile pass for that tick (log + error code, per §4) rather than falling through to the diff, or every transient API-server hiccup becomes a false mass-delete storm published to Server. This is the single most dangerous failure mode in the design and must be a hard early-return, tested explicitly (§9). - **Stale reconstructed object on missed-delete.** The synthesized Delete event (§3(c)) carries the last-known Store copy, not a live cluster read (the object no longer exists to read) — downstream consumers relying on delete-event payload freshness (there are none identified in Server's `Delete` handling, which only uses it for a keyed row delete, `models/meshsync_events.go:242-246`) are unaffected, but this should be called out in the PR description as an intentional, unavoidable characteristic. diff --git a/main.go b/main.go index e15691d9..0d4b992d 100644 --- a/main.go +++ b/main.go @@ -94,14 +94,14 @@ func parseFlags() { &outputNamespacesString, "outputNamespaces", "", - "k8s namespaces for which limit the output, comma separated list f.e. \"default,agile-otter\", applicable for both nats and file output mode", + "k8s namespaces for which limit the output, comma separated list f.e. \"default,agile-otter\", applicable for both broker and file output mode", ) var outputResourcesString string flag.StringVar( &outputResourcesString, "outputResources", "", - "k8s resources for which limit the output, comma separated case insensitive list of k8s resources, f.e. \"pod,deployment,service\", applicable for both nats and file output mode", + "k8s resources for which limit the output, comma separated case insensitive list of k8s resources, f.e. \"pod,deployment,service\", applicable for both broker and file output mode", ) flag.DurationVar( &stopAfterDuration, diff --git a/pkg/lib/meshsync/meshsync.go b/pkg/lib/meshsync/meshsync.go index bb11ad4d..7e6fa4b0 100644 --- a/pkg/lib/meshsync/meshsync.go +++ b/pkg/lib/meshsync/meshsync.go @@ -370,8 +370,8 @@ func determineUseCRDFlag( log logger.Handler, kubeClient *mesherykube.Client, ) bool { - // if output mode is not nats generally it is not expected to have CRD present in cluster. - // theoretically CRD could be present even in file, channel output mode. + // if output mode is not broker generally it is not expected to have CRD present in cluster. + // theoretically CRD could be present even in file output mode. // hence check if CRD are present in the cluster, // and only skip them if it is not present. crd, errGetMeshsyncCRD := config.GetMeshsyncCRD(kubeClient.DynamicKubeClient)