From 1d38665ddcf21526eb6c407339c39679a3b34121 Mon Sep 17 00:00:00 2001
From: Lee Calcote
Date: Sat, 4 Jul 2026 19:33:30 -0500
Subject: [PATCH 1/2] docs: correct stale output-mode and informer-filtering
claims; drop stray manifests
- README named the default output mode "nats"; the -output flag value is
"broker" (internal/config/types.go). Rename and clarify that broker mode
publishes to Meshery Broker over NATS.
- docs/agent-instructions/architecture.md claimed the dynamic informer
factory filters via a label selector derived from the informer blacklist;
GetListOptionsFunc is a deliberate no-op and filtering happens in the
watch-list config (internal/config/crd_config.go).
- Remove cert-manager pod dumps accidentally committed at the repo root
(all-certmgs.yaml, cert-manager-cainjector-*.yaml,
cert-manager-webhook-*.yaml); referenced nowhere.
Signed-off-by: Lee Calcote
---
README.md | 2 +-
all-certmgs.yaml | 0
cert-manager-cainjector-c7d4dbdd9-jlstt.yaml | 139 ---------------
cert-manager-webhook-847d7676c9-89rtq.yaml | 171 -------------------
docs/agent-instructions/architecture.md | 2 +-
5 files changed, 2 insertions(+), 312 deletions(-)
delete mode 100644 all-certmgs.yaml
delete mode 100644 cert-manager-cainjector-c7d4dbdd9-jlstt.yaml
delete mode 100644 cert-manager-webhook-847d7676c9-89rtq.yaml
diff --git a/README.md b/README.md
index bb92e530..1c216cb9 100644
--- a/README.md
+++ b/README.md
@@ -37,7 +37,7 @@ alt="Meshery Logo" width="70%" />
MeshSync is Meshery's event-driven, continuous discovery and synchronization engine. It ensures that the configuration and operational state of Kubernetes (and any supported Meshery platform) are known to Meshery Server. When deployed into a Kubernetes cluster, MeshSync runs as a custom controller under the control of [Meshery Operator](https://docs.meshery.io/concepts/architecture/operator) and publishes resource changes over Meshery Broker (NATS).
-MeshSync runs in one of two modes: **nats** (default - publishes Kubernetes resource events to NATS) and **file** (writes deduplicated cluster snapshots to disk, with no NATS or CRD dependency). Run `meshsync --help` for input parameters.
+MeshSync runs in one of two output modes: **broker** (default - publishes Kubernetes resource events to Meshery Broker over NATS) and **file** (writes deduplicated cluster snapshots to disk, with no NATS or CRD dependency). Run `meshsync --help` for input parameters.
## Documentation
diff --git a/all-certmgs.yaml b/all-certmgs.yaml
deleted file mode 100644
index e69de29b..00000000
diff --git a/cert-manager-cainjector-c7d4dbdd9-jlstt.yaml b/cert-manager-cainjector-c7d4dbdd9-jlstt.yaml
deleted file mode 100644
index 723375ba..00000000
--- a/cert-manager-cainjector-c7d4dbdd9-jlstt.yaml
+++ /dev/null
@@ -1,139 +0,0 @@
-apiVersion: v1
-kind: Pod
-metadata:
- creationTimestamp: "2024-04-01T04:29:32Z"
- generateName: cert-manager-cainjector-c7d4dbdd9-
- labels:
- app: cainjector
- app.kubernetes.io/component: cainjector
- app.kubernetes.io/instance: cert-manager
- app.kubernetes.io/managed-by: Helm
- app.kubernetes.io/name: cainjector
- app.kubernetes.io/version: v1.14.4
- helm.sh/chart: cert-manager-v1.14.4
- pod-template-hash: c7d4dbdd9
- name: cert-manager-cainjector-c7d4dbdd9-jlstt
- namespace: cert-manager
- ownerReferences:
- - apiVersion: apps/v1
- blockOwnerDeletion: true
- controller: true
- kind: ReplicaSet
- name: cert-manager-cainjector-c7d4dbdd9
- uid: 4ad5f4b4-6ddc-4bad-8208-418435b3b8d7
- resourceVersion: "4476"
- uid: b905aeb8-3633-4ca9-b350-df3a87cdd60e
-spec:
- containers:
- - args:
- - --v=2
- - --leader-election-namespace=kube-system
- env:
- - name: POD_NAMESPACE
- valueFrom:
- fieldRef:
- apiVersion: v1
- fieldPath: metadata.namespace
- image: quay.io/jetstack/cert-manager-cainjector:v1.14.4
- imagePullPolicy: IfNotPresent
- name: cert-manager-cainjector
- resources: {}
- securityContext:
- allowPrivilegeEscalation: false
- capabilities:
- drop:
- - ALL
- readOnlyRootFilesystem: true
- terminationMessagePath: /dev/termination-log
- terminationMessagePolicy: File
- volumeMounts:
- - mountPath: /var/run/secrets/kubernetes.io/serviceaccount
- name: kube-api-access-sktmn
- readOnly: true
- dnsPolicy: ClusterFirst
- enableServiceLinks: false
- nodeName: c3-medium-x86-03-meshery
- nodeSelector:
- kubernetes.io/os: linux
- preemptionPolicy: PreemptLowerPriority
- priority: 0
- restartPolicy: Always
- schedulerName: default-scheduler
- securityContext:
- runAsNonRoot: true
- seccompProfile:
- type: RuntimeDefault
- serviceAccount: cert-manager-cainjector
- serviceAccountName: cert-manager-cainjector
- terminationGracePeriodSeconds: 30
- tolerations:
- - effect: NoExecute
- key: node.kubernetes.io/not-ready
- operator: Exists
- tolerationSeconds: 300
- - effect: NoExecute
- key: node.kubernetes.io/unreachable
- operator: Exists
- tolerationSeconds: 300
- volumes:
- - name: kube-api-access-sktmn
- projected:
- defaultMode: 420
- sources:
- - serviceAccountToken:
- expirationSeconds: 3607
- path: token
- - configMap:
- items:
- - key: ca.crt
- path: ca.crt
- name: kube-root-ca.crt
- - downwardAPI:
- items:
- - fieldRef:
- apiVersion: v1
- fieldPath: metadata.namespace
- path: namespace
-status:
- conditions:
- - lastProbeTime: null
- lastTransitionTime: "2024-04-01T04:29:34Z"
- status: "True"
- type: PodReadyToStartContainers
- - lastProbeTime: null
- lastTransitionTime: "2024-04-01T04:29:32Z"
- status: "True"
- type: Initialized
- - lastProbeTime: null
- lastTransitionTime: "2024-04-01T04:29:34Z"
- status: "True"
- type: Ready
- - lastProbeTime: null
- lastTransitionTime: "2024-04-01T04:29:34Z"
- status: "True"
- type: ContainersReady
- - lastProbeTime: null
- lastTransitionTime: "2024-04-01T04:29:32Z"
- status: "True"
- type: PodScheduled
- containerStatuses:
- - containerID: containerd://1c7b0a11d4c38a0fce9c3abf2b69358eae486bf383344a18b730db6bf5e135a4
- image: quay.io/jetstack/cert-manager-cainjector:v1.14.4
- imageID: quay.io/jetstack/cert-manager-cainjector@sha256:30286297e5b4b71a86759d297a8109c6a1649fdc68d28f618d87edf12a2da417
- lastState: {}
- name: cert-manager-cainjector
- ready: true
- restartCount: 0
- started: true
- state:
- running:
- startedAt: "2024-04-01T04:29:33Z"
- hostIP: 139.178.83.85
- hostIPs:
- - ip: 139.178.83.85
- phase: Running
- podIP: 192.168.0.14
- podIPs:
- - ip: 192.168.0.14
- qosClass: BestEffort
- startTime: "2024-04-01T04:29:32Z"
diff --git a/cert-manager-webhook-847d7676c9-89rtq.yaml b/cert-manager-webhook-847d7676c9-89rtq.yaml
deleted file mode 100644
index 57baf1e0..00000000
--- a/cert-manager-webhook-847d7676c9-89rtq.yaml
+++ /dev/null
@@ -1,171 +0,0 @@
-apiVersion: v1
-kind: Pod
-metadata:
- creationTimestamp: "2024-04-01T04:29:32Z"
- generateName: cert-manager-webhook-847d7676c9-
- labels:
- app: webhook
- app.kubernetes.io/component: webhook
- app.kubernetes.io/instance: cert-manager
- app.kubernetes.io/managed-by: Helm
- app.kubernetes.io/name: webhook
- app.kubernetes.io/version: v1.14.4
- helm.sh/chart: cert-manager-v1.14.4
- pod-template-hash: 847d7676c9
- name: cert-manager-webhook-847d7676c9-89rtq
- namespace: cert-manager
- ownerReferences:
- - apiVersion: apps/v1
- blockOwnerDeletion: true
- controller: true
- kind: ReplicaSet
- name: cert-manager-webhook-847d7676c9
- uid: 9ca18977-52e3-474e-93e7-cbcaba9a13fa
- resourceVersion: "47870864"
- uid: 4b127dff-feeb-4962-81b2-27b87d2a1418
-spec:
- containers:
- - args:
- - --v=2
- - --secure-port=10250
- - --dynamic-serving-ca-secret-namespace=$(POD_NAMESPACE)
- - --dynamic-serving-ca-secret-name=cert-manager-webhook-ca
- - --dynamic-serving-dns-names=cert-manager-webhook
- - --dynamic-serving-dns-names=cert-manager-webhook.$(POD_NAMESPACE)
- - --dynamic-serving-dns-names=cert-manager-webhook.$(POD_NAMESPACE).svc
- env:
- - name: POD_NAMESPACE
- valueFrom:
- fieldRef:
- apiVersion: v1
- fieldPath: metadata.namespace
- image: quay.io/jetstack/cert-manager-webhook:v1.14.4
- imagePullPolicy: IfNotPresent
- livenessProbe:
- failureThreshold: 3
- httpGet:
- path: /livez
- port: 6080
- scheme: HTTP
- initialDelaySeconds: 60
- periodSeconds: 10
- successThreshold: 1
- timeoutSeconds: 1
- name: cert-manager-webhook
- ports:
- - containerPort: 10250
- name: https
- protocol: TCP
- - containerPort: 6080
- name: healthcheck
- protocol: TCP
- readinessProbe:
- failureThreshold: 3
- httpGet:
- path: /healthz
- port: 6080
- scheme: HTTP
- initialDelaySeconds: 5
- periodSeconds: 5
- successThreshold: 1
- timeoutSeconds: 1
- resources: {}
- securityContext:
- allowPrivilegeEscalation: false
- capabilities:
- drop:
- - ALL
- readOnlyRootFilesystem: true
- terminationMessagePath: /dev/termination-log
- terminationMessagePolicy: File
- volumeMounts:
- - mountPath: /var/run/secrets/kubernetes.io/serviceaccount
- name: kube-api-access-9657m
- readOnly: true
- dnsPolicy: ClusterFirst
- enableServiceLinks: false
- nodeName: c3-medium-x86-03-meshery
- nodeSelector:
- kubernetes.io/os: linux
- preemptionPolicy: PreemptLowerPriority
- priority: 0
- restartPolicy: Always
- schedulerName: default-scheduler
- securityContext:
- runAsNonRoot: true
- seccompProfile:
- type: RuntimeDefault
- serviceAccount: cert-manager-webhook
- serviceAccountName: cert-manager-webhook
- terminationGracePeriodSeconds: 30
- tolerations:
- - effect: NoExecute
- key: node.kubernetes.io/not-ready
- operator: Exists
- tolerationSeconds: 300
- - effect: NoExecute
- key: node.kubernetes.io/unreachable
- operator: Exists
- tolerationSeconds: 300
- volumes:
- - name: kube-api-access-9657m
- projected:
- defaultMode: 420
- sources:
- - serviceAccountToken:
- expirationSeconds: 3607
- path: token
- - configMap:
- items:
- - key: ca.crt
- path: ca.crt
- name: kube-root-ca.crt
- - downwardAPI:
- items:
- - fieldRef:
- apiVersion: v1
- fieldPath: metadata.namespace
- path: namespace
-status:
- conditions:
- - lastProbeTime: null
- lastTransitionTime: "2024-04-01T04:29:34Z"
- status: "True"
- type: PodReadyToStartContainers
- - lastProbeTime: null
- lastTransitionTime: "2024-04-01T04:29:32Z"
- status: "True"
- type: Initialized
- - lastProbeTime: null
- lastTransitionTime: "2024-11-06T05:50:58Z"
- status: "True"
- type: Ready
- - lastProbeTime: null
- lastTransitionTime: "2024-11-06T05:50:58Z"
- status: "True"
- type: ContainersReady
- - lastProbeTime: null
- lastTransitionTime: "2024-04-01T04:29:32Z"
- status: "True"
- type: PodScheduled
- containerStatuses:
- - containerID: containerd://f60e25f3ca9dddc430ade42179a757868cf2264e52682a003fb2fd52df7b8efc
- image: quay.io/jetstack/cert-manager-webhook:v1.14.4
- imageID: quay.io/jetstack/cert-manager-webhook@sha256:11f7e7c462da3c0329e0a1e695a7bd37d6b3c28312d4edd4cc8d36f70ecbfa63
- lastState: {}
- name: cert-manager-webhook
- ready: true
- restartCount: 0
- started: true
- state:
- running:
- startedAt: "2024-04-01T04:29:33Z"
- hostIP: 139.178.83.85
- hostIPs:
- - ip: 139.178.83.85
- phase: Running
- podIP: 192.168.0.15
- podIPs:
- - ip: 192.168.0.15
- qosClass: BestEffort
- startTime: "2024-04-01T04:29:32Z"
diff --git a/docs/agent-instructions/architecture.md b/docs/agent-instructions/architecture.md
index 6d363cd8..891111c0 100644
--- a/docs/agent-instructions/architecture.md
+++ b/docs/agent-instructions/architecture.md
@@ -29,7 +29,7 @@ main.go --parses CLI flags--> pkg/lib/meshsync.Run(...)
- `main.go` parses flags (`-output`, `-outputFile`, `-outputNamespaces`, `-outputResources`, `-stopAfter`) and calls `pkg/lib/meshsync.Run(...)`.
- `meshsync.Handler` (`meshsync/meshsync.go`) holds the config, logger, broker handle, dynamic informer factory, kube client, channel pool, output writer, and output-filtration config. `meshsync.New(...)` wires them together and derives the cluster ID via `pkg/utils.GetClusterID`.
-- `GetDynamicInformer` builds a `dynamicinformer.DynamicSharedInformerFactory` filtered by a label selector derived from the config's informer blacklist.
+- `GetDynamicInformer` builds a `dynamicinformer.DynamicSharedInformerFactory`. Resource filtering happens in the watch-list config (`internal/config/crd_config.go` decides which informers get registered); the factory's list-options hook (`GetListOptionsFunc`) is a deliberate no-op.
## Discovery Pipeline (`internal/pipeline`)
From ed838612e6d89072513b4a6fa717ed7612cf5de7 Mon Sep 17 00:00:00 2001
From: Lee Calcote
Date: Sat, 4 Jul 2026 19:52:16 -0500
Subject: [PATCH 2/2] docs: sweep remaining stale output-mode and list-options
references
Review follow-up: align the -outputNamespaces/-outputResources flag help
text and the determineUseCRDFlag comment with the actual broker output
mode (and drop the reference to a nonexistent channel output mode), and
correct fd5-periodic-reconciliation.md where it still described
GetListOptionsFunc as a blacklist label selector.
Signed-off-by: Lee Calcote
---
docs/design/fd5-periodic-reconciliation.md | 4 ++--
main.go | 4 ++--
pkg/lib/meshsync/meshsync.go | 4 ++--
3 files changed, 6 insertions(+), 6 deletions(-)
diff --git a/docs/design/fd5-periodic-reconciliation.md b/docs/design/fd5-periodic-reconciliation.md
index 9ecd4ee0..f8bb175f 100644
--- a/docs/design/fd5-periodic-reconciliation.md
+++ b/docs/design/fd5-periodic-reconciliation.md
@@ -107,7 +107,7 @@ Implemented as the third branch above: for each reconciled GVR, take the key set
- **`meshsync/meshsync.go`**: `Handler` gets no new fields — a reconcile loop is a *behavior* of `Run()`, not new handler state, keeping with `pipeline.New`'s per-call-fresh philosophy the architecture doc already mandates.
- **`meshsync/handlers.go`** — the core of the feature:
- **New unexported function `(h *Handler) reconcileLoop(intervalStr string, stopCh channels.StopChannel)`**, started as a goroutine from `Run()` (alongside the existing `startAndTrackDiscovery()` call, `handlers.go:57-58`) only when the resolved interval is non-zero. On each tick (a `time.Ticker`, jittered per §"scale" below), it iterates `h.stores` (the same map `startDiscovery` populates at `discovery.go:48` — `h.stores[gvr-name] = cache.Store`) and, for each GVR present, performs:
- 1. A fresh, filtered LIST against `h.kubeClient.DynamicKubeClient` for that GVR (respecting the same `GetListOptionsFunc` blacklist label selector already used for informer construction, `meshsync/meshsync.go:34-53`, so reconcile never re-lists resource kinds the informer itself was configured to exclude).
+ 1. A fresh, filtered LIST against `h.kubeClient.DynamicKubeClient` for that GVR (reconcile iterates only the GVRs whose informer pipelines were registered from the watch-list config; `GetListOptionsFunc` is a deliberate no-op, so there is no list-time label filtering to rely on).
2. Diff against `store.List()`/`store.ListKeys()` per §3(b)/(c).
3. Publish only drifted objects through the *existing* `internal/pipeline.RegisterInformer.publishItem`-equivalent path — since `publishItem` is a method on `RegisterInformer` (pipeline-internal, not exported), the cleanest reuse is to export a small helper in `internal/pipeline` (e.g. `pipeline.PublishDrift(log, outputWriter, clusterID, outputFiltration, cfg PipelineConfig, obj *unstructured.Unstructured, evtype broker.EventType) error`) that wraps the existing `model.ParseList` + `checkMustSkip` + `outputWriter.Write` sequence (lines 93-111 of `internal/pipeline/handlers.go`), called from both `RegisterInformer.publishItem` (refactored to delegate to it) and the new reconcile loop — avoiding duplicating the skip/filter/write logic in two places, which the CLAUDE.md "consistency with existing patterns... migrate the existing code accordingly" directive requires rather than a second parallel implementation.
- Guard the whole loop on the `channels.Stop` channel exactly like `WatchCRDs` does (`handlers.go:359-361`) for clean shutdown.
@@ -151,7 +151,7 @@ The default-off interval (§3(d)) **is** the feature flag — no separate boolea
## 8. Risks, Failure Modes, Perf/Scale
-- **Re-list storm on large clusters.** A naive "list every configured GVR every tick" is the primary named risk. Mitigations designed in: (1) default-off (§3(d)); (2) enforced floor interval (5m minimum, §3(d)); (3) intra-tick staggering across GVRs with bounded concurrency (§4); (4) fleet-wide jitter via the existing `jitter()` helper (§4) so many clusters' MeshSync instances don't synchronize; (5) reconcile scope is **exactly** the informer's existing configured GVR set filtered through `GetListOptionsFunc`'s blacklist label selector (§4) — it never lists more than what's already being watched, and it respects `-outputResources` (via `outputFiltration.ResourceSet`, already applied at the pipeline-construction/write layer) so a restricted deployment reconciles only its restricted resource set. **`-outputNamespaces` is a write-time filter, not an informer/LIST-time scope** (§2) — a genuinely namespace-scoped reconcile (listing only within the allowed namespaces at the API-server level, cheaper than a cluster-wide LIST filtered client-side) is not achievable without deeper informer-factory changes (`NewFilteredDynamicSharedInformerFactory` takes one cluster-wide-or-single-namespace argument, not a set) — flagged explicitly as an out-of-scope enhancement in §11, with the interim mitigation being that write-time filtering still means a namespace-restricted deployment publishes no more drift than it would publish live events, even though its LIST cost is unreduced.
+- **Re-list storm on large clusters.** A naive "list every configured GVR every tick" is the primary named risk. Mitigations designed in: (1) default-off (§3(d)); (2) enforced floor interval (5m minimum, §3(d)); (3) intra-tick staggering across GVRs with bounded concurrency (§4); (4) fleet-wide jitter via the existing `jitter()` helper (§4) so many clusters' MeshSync instances don't synchronize; (5) reconcile scope is **exactly** the informer's existing configured GVR set as registered from the watch-list config (§4; `GetListOptionsFunc` is a deliberate no-op) — it never lists more than what's already being watched, and it respects `-outputResources` (via `outputFiltration.ResourceSet`, already applied at the pipeline-construction/write layer) so a restricted deployment reconciles only its restricted resource set. **`-outputNamespaces` is a write-time filter, not an informer/LIST-time scope** (§2) — a genuinely namespace-scoped reconcile (listing only within the allowed namespaces at the API-server level, cheaper than a cluster-wide LIST filtered client-side) is not achievable without deeper informer-factory changes (`NewFilteredDynamicSharedInformerFactory` takes one cluster-wide-or-single-namespace argument, not a set) — flagged explicitly as an out-of-scope enhancement in §11, with the interim mitigation being that write-time filtering still means a namespace-restricted deployment publishes no more drift than it would publish live events, even though its LIST cost is unreduced.
- **Dedup/RV-suppression interaction failing silently.** If the reconcile loop's own RV-diff (§3(b)) has a bug and always treats "unchanged" as "changed," it defeats the entire purpose and re-floods Server every tick. Test plan (§9) includes an explicit "steady state produces zero publishes" assertion, not just a "drift produces N publishes" assertion — the null case is equally load-bearing.
- **Missed-delete false positives from partial/paginated or rate-limited LIST failures.** A LIST that fails midway or is truncated must never be diffed as "everything not returned is deleted" — a failed/partial LIST must abort that GVR's reconcile pass for that tick (log + error code, per §4) rather than falling through to the diff, or every transient API-server hiccup becomes a false mass-delete storm published to Server. This is the single most dangerous failure mode in the design and must be a hard early-return, tested explicitly (§9).
- **Stale reconstructed object on missed-delete.** The synthesized Delete event (§3(c)) carries the last-known Store copy, not a live cluster read (the object no longer exists to read) — downstream consumers relying on delete-event payload freshness (there are none identified in Server's `Delete` handling, which only uses it for a keyed row delete, `models/meshsync_events.go:242-246`) are unaffected, but this should be called out in the PR description as an intentional, unavoidable characteristic.
diff --git a/main.go b/main.go
index e15691d9..0d4b992d 100644
--- a/main.go
+++ b/main.go
@@ -94,14 +94,14 @@ func parseFlags() {
&outputNamespacesString,
"outputNamespaces",
"",
- "k8s namespaces for which limit the output, comma separated list f.e. \"default,agile-otter\", applicable for both nats and file output mode",
+ "k8s namespaces for which limit the output, comma separated list f.e. \"default,agile-otter\", applicable for both broker and file output mode",
)
var outputResourcesString string
flag.StringVar(
&outputResourcesString,
"outputResources",
"",
- "k8s resources for which limit the output, comma separated case insensitive list of k8s resources, f.e. \"pod,deployment,service\", applicable for both nats and file output mode",
+ "k8s resources for which limit the output, comma separated case insensitive list of k8s resources, f.e. \"pod,deployment,service\", applicable for both broker and file output mode",
)
flag.DurationVar(
&stopAfterDuration,
diff --git a/pkg/lib/meshsync/meshsync.go b/pkg/lib/meshsync/meshsync.go
index bb11ad4d..7e6fa4b0 100644
--- a/pkg/lib/meshsync/meshsync.go
+++ b/pkg/lib/meshsync/meshsync.go
@@ -370,8 +370,8 @@ func determineUseCRDFlag(
log logger.Handler,
kubeClient *mesherykube.Client,
) bool {
- // if output mode is not nats generally it is not expected to have CRD present in cluster.
- // theoretically CRD could be present even in file, channel output mode.
+ // if output mode is not broker generally it is not expected to have CRD present in cluster.
+ // theoretically CRD could be present even in file output mode.
// hence check if CRD are present in the cluster,
// and only skip them if it is not present.
crd, errGetMeshsyncCRD := config.GetMeshsyncCRD(kubeClient.DynamicKubeClient)